diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index b09335c80..cc00a131a 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -8,7 +8,7 @@ "name": "mem0", "source": { "source": "local", - "path": "./integrations/mem0-plugin" + "path": "./integrations/codex-plugin" }, "policy": { "installation": "AVAILABLE", diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 392e9c6f5..d46a19e49 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -12,7 +12,7 @@ "name": "mem0", "source": "./integrations/claude-code-plugin", "description": "Cross-session memory and token savings for coding agents.", - "version": "0.3.0" + "version": "0.3.1" } ] } diff --git a/.codex-plugin/marketplace.json b/.codex-plugin/marketplace.json index b09335c80..cc00a131a 100644 --- a/.codex-plugin/marketplace.json +++ b/.codex-plugin/marketplace.json @@ -8,7 +8,7 @@ "name": "mem0", "source": { "source": "local", - "path": "./integrations/mem0-plugin" + "path": "./integrations/codex-plugin" }, "policy": { "installation": "AVAILABLE", diff --git a/.cursor-plugin/marketplace.json b/.cursor-plugin/marketplace.json index f931f04ff..409b7d8fc 100644 --- a/.cursor-plugin/marketplace.json +++ b/.cursor-plugin/marketplace.json @@ -10,9 +10,9 @@ "plugins": [ { "name": "mem0", - "source": "./integrations/mem0-plugin", - "description": "Mem0 memory layer for AI applications. Add persistent memory, personalization, and semantic search.", - "version": "0.2.15" + "source": "./integrations/cursor-plugin", + "description": "Cross-session memory and token savings for coding agents.", + "version": "0.3.1" } ] } diff --git a/.github/AGENTS.md b/.github/AGENTS.md index e207cb9fd..573480da5 100644 --- a/.github/AGENTS.md +++ b/.github/AGENTS.md @@ -18,9 +18,9 @@ Package workflows keep their own push-to-main and manual triggers. Their `pull_r | Python CLI | `cli-python-ci.yml` | Push to main (`cli/python/`), manual | Ruff + pytest + hatch build on Python 3.10, 3.11, 3.12 | | Node CLI | `cli-node-ci.yml` | Push to main (`cli/node/`), manual | Biome + tsc + vitest + tsup on Node 20, 22 | | OpenClaw | `openclaw-checks.yml` | Push to main (`integrations/openclaw/`), manual | tsc + vitest (Codecov) + tsup on Node 20, 22 | -| Mem0 Plugin (legacy) | `mem0-plugin-checks.yml` | Push to main (`integrations/mem0-plugin/`, excluding `.opencode-plugin/`), manual | pytest + hook exec bits + JSON manifest validation on Python 3.10, 3.11, 3.12 | -| Claude Code Plugin | `claude-code-plugin-checks.yml` | Push to main (`integrations/claude-code-plugin/`), manual | pytest + ruff + JSON manifest validation on Python 3.10, 3.11, 3.12 | -| OpenCode Plugin | `opencode-plugin-checks.yml` | Push to main (`.opencode-plugin/`), manual | Bun: tsc + build + dist artifact check | +| Agent Plugins Python | `agent-plugins-python-checks.yml` | Push to main (shared Python core and native/portable plugin directories), manual | Runtime tests on Python 3.10; full pytest on 3.11, 3.12; ruff + generated-package drift on 3.12 | +| Agent Plugins TypeScript | `agent-plugins-typescript-checks.yml` | Push to main (`integrations/agent-plugin-core/typescript/`), manual | tsc + node:test on Node 22 | +| OpenCode Plugin | `opencode-plugin-checks.yml` | Push to main (`integrations/opencode-plugin/`), manual | Bun: tsc + build + dist artifact check | | Pi Agent Plugin | `pi-agent-plugin-checks.yml` | Push to main (`integrations/pi-agent-plugin/`), manual | tsc + vitest + tsup on Node 20, 22 | | DeepSeek Harness Plugin | `deepseek-plugin-checks.yml` | Push to main (`integrations/deepseek-plugin/`), manual | tsc + vitest + tsup on Node 20, 22 | | n8n Node | `n8n-nodes-mem0-checks.yml` | Push to main (`integrations/n8n-nodes-mem0/`), manual | ESLint + tsc build on Node 20 | diff --git a/.github/workflows/agent-plugins-python-checks.yml b/.github/workflows/agent-plugins-python-checks.yml new file mode 100644 index 000000000..dde6598ab --- /dev/null +++ b/.github/workflows/agent-plugins-python-checks.yml @@ -0,0 +1,97 @@ +name: Agent Plugins Python Checks + +# Python runtime, adapters, generated bundles, and portable plugin validation. +# On PRs this is invoked by ci-gate.yml (the single required check); +# push-to-main and manual runs remain standalone. +on: + workflow_dispatch: + push: + branches: [main] + paths: + - 'integrations/agent-plugin-core/**' + - '!integrations/agent-plugin-core/typescript/**' + - 'integrations/mem0-agent-plugin/**' + - 'integrations/claude-code-plugin/**' + - 'integrations/cursor-plugin/**' + - 'integrations/codex-plugin/**' + - 'integrations/kimi-plugin/**' + - 'integrations/antigravity-plugin/**' + - 'marketplace.json' + - '.agents/plugins/marketplace.json' + - '.claude-plugin/marketplace.json' + - '.codex-plugin/marketplace.json' + - '.cursor-plugin/marketplace.json' + - '.kimi-plugin/marketplace.json' + - '.github/workflows/agent-plugins-python-checks.yml' + workflow_call: + +jobs: + test: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + python-version: ["3.10", "3.11", "3.12"] + steps: + - uses: actions/checkout@v4 + + - name: Set up Python ${{ matrix.python-version }} + uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + + - name: Install runtime test tooling + if: matrix.python-version == '3.10' + run: pip install pytest + + - name: Install build and test tooling + if: matrix.python-version != '3.10' + run: pip install pytest ruff -r integrations/agent-plugin-core/requirements-dev.txt + + - name: Check Python runtime compatibility + run: >- + python3 -m compileall -q + integrations/agent-plugin-core/python + integrations/claude-code-plugin/adapters + integrations/cursor-plugin/hooks + integrations/codex-plugin/hooks + integrations/kimi-plugin/hooks + integrations/antigravity-plugin/hooks + + - name: Lint + if: matrix.python-version == '3.12' + run: >- + python3 -m ruff check + integrations/agent-plugin-core + integrations/claude-code-plugin + integrations/cursor-plugin + integrations/codex-plugin + integrations/kimi-plugin + integrations/antigravity-plugin + + - name: Verify installable plugins are current + if: matrix.python-version == '3.12' + run: | + for host in claude-code cursor codex kimi antigravity; do + python3 integrations/agent-plugin-core/build/build.py "$host" --kind native --check + done + python3 integrations/agent-plugin-core/build/build.py mem0-agent-plugin --kind portable --check + + - name: Run Python 3.10 runtime tests + if: matrix.python-version == '3.10' + run: >- + python3 -m pytest -q + integrations/claude-code-plugin/tests/test_memory_core.py + integrations/claude-code-plugin/tests/test_telemetry.py + + - name: Run full tests + if: matrix.python-version != '3.10' + run: >- + python3 -m pytest -q + integrations/agent-plugin-core/tests + integrations/claude-code-plugin/tests + integrations/cursor-plugin/tests + integrations/codex-plugin/tests + integrations/kimi-plugin/tests + integrations/antigravity-plugin/tests + --ignore=integrations/claude-code-plugin/tests/integration diff --git a/.github/workflows/agent-plugins-typescript-checks.yml b/.github/workflows/agent-plugins-typescript-checks.yml new file mode 100644 index 000000000..1b480edbd --- /dev/null +++ b/.github/workflows/agent-plugins-typescript-checks.yml @@ -0,0 +1,42 @@ +name: Agent Plugins TypeScript Checks + +# Shared TypeScript runtime checks. Each consuming integration keeps its own +# build workflow, which is also triggered when this shared core changes. +on: + workflow_dispatch: + push: + branches: [main] + paths: + - 'integrations/agent-plugin-core/typescript/**' + - '.github/workflows/agent-plugins-typescript-checks.yml' + workflow_call: + +jobs: + test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Install pnpm + uses: pnpm/action-setup@v4 + with: + version: 10 + + - name: Set up Node.js + uses: actions/setup-node@v4 + with: + node-version: 22 + cache: 'pnpm' + cache-dependency-path: integrations/agent-plugin-core/typescript/pnpm-lock.yaml + + - name: Install dependencies + working-directory: integrations/agent-plugin-core/typescript + run: pnpm install --frozen-lockfile + + - name: Type check + working-directory: integrations/agent-plugin-core/typescript + run: pnpm typecheck + + - name: Run tests + working-directory: integrations/agent-plugin-core/typescript + run: pnpm test diff --git a/.github/workflows/ci-gate.yml b/.github/workflows/ci-gate.yml index 2359de80b..b4fd6dc81 100644 --- a/.github/workflows/ci-gate.yml +++ b/.github/workflows/ci-gate.yml @@ -38,8 +38,8 @@ jobs: cli_python: ${{ steps.filter.outputs.cli_python }} cli_node: ${{ steps.filter.outputs.cli_node }} openclaw: ${{ steps.filter.outputs.openclaw }} - mem0_plugin: ${{ steps.filter.outputs.mem0_plugin }} - claude_code_plugin: ${{ steps.filter.outputs.claude_code_plugin }} + agent_plugins_python: ${{ steps.filter.outputs.agent_plugins_python }} + agent_plugins_typescript: ${{ steps.filter.outputs.agent_plugins_typescript }} opencode_plugin: ${{ steps.filter.outputs.opencode_plugin }} pi_agent_plugin: ${{ steps.filter.outputs.pi_agent_plugin }} deepseek_plugin: ${{ steps.filter.outputs.deepseek_plugin }} @@ -76,27 +76,43 @@ jobs: - '.github/workflows/ci-gate.yml' openclaw: - 'integrations/openclaw/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/openclaw-checks.yml' - '.github/workflows/ci-gate.yml' - mem0_plugin: - - 'integrations/mem0-plugin/**' - - '!integrations/mem0-plugin/.opencode-plugin/**' - - '.github/workflows/mem0-plugin-checks.yml' - - '.github/workflows/ci-gate.yml' - claude_code_plugin: + agent_plugins_python: + - 'integrations/agent-plugin-core/**' + - '!integrations/agent-plugin-core/typescript/**' + - 'integrations/mem0-agent-plugin/**' - 'integrations/claude-code-plugin/**' - - '.github/workflows/claude-code-plugin-checks.yml' + - 'integrations/cursor-plugin/**' + - 'integrations/codex-plugin/**' + - 'integrations/kimi-plugin/**' + - 'integrations/antigravity-plugin/**' + - 'marketplace.json' + - '.agents/plugins/marketplace.json' + - '.claude-plugin/marketplace.json' + - '.codex-plugin/marketplace.json' + - '.cursor-plugin/marketplace.json' + - '.kimi-plugin/marketplace.json' + - '.github/workflows/agent-plugins-python-checks.yml' + - '.github/workflows/ci-gate.yml' + agent_plugins_typescript: + - 'integrations/agent-plugin-core/typescript/**' + - '.github/workflows/agent-plugins-typescript-checks.yml' - '.github/workflows/ci-gate.yml' opencode_plugin: - - 'integrations/mem0-plugin/.opencode-plugin/**' + - 'integrations/opencode-plugin/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/opencode-plugin-checks.yml' - '.github/workflows/ci-gate.yml' pi_agent_plugin: - 'integrations/pi-agent-plugin/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/pi-agent-plugin-checks.yml' - '.github/workflows/ci-gate.yml' deepseek_plugin: - 'integrations/deepseek-plugin/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/deepseek-plugin-checks.yml' - '.github/workflows/ci-gate.yml' n8n_nodes_mem0: @@ -160,18 +176,17 @@ jobs: uses: ./.github/workflows/openclaw-checks.yml secrets: inherit - mem0-plugin: - name: Mem0 Plugin + agent-plugins-python: + name: Agent Plugins Python needs: changes - if: needs.changes.outputs.mem0_plugin == 'true' - uses: ./.github/workflows/mem0-plugin-checks.yml - secrets: inherit + if: needs.changes.outputs.agent_plugins_python == 'true' + uses: ./.github/workflows/agent-plugins-python-checks.yml - claude-code-plugin: - name: Claude Code Plugin + agent-plugins-typescript: + name: Agent Plugins TypeScript needs: changes - if: needs.changes.outputs.claude_code_plugin == 'true' - uses: ./.github/workflows/claude-code-plugin-checks.yml + if: needs.changes.outputs.agent_plugins_typescript == 'true' + uses: ./.github/workflows/agent-plugins-typescript-checks.yml secrets: inherit opencode-plugin: @@ -248,8 +263,8 @@ jobs: - cli-python - cli-node - openclaw - - mem0-plugin - - claude-code-plugin + - agent-plugins-python + - agent-plugins-typescript - opencode-plugin - pi-agent-plugin - deepseek-plugin diff --git a/.github/workflows/claude-code-plugin-checks.yml b/.github/workflows/claude-code-plugin-checks.yml deleted file mode 100644 index 01d0dde8e..000000000 --- a/.github/workflows/claude-code-plugin-checks.yml +++ /dev/null @@ -1,46 +0,0 @@ -name: Claude Code Plugin Checks - -# On PRs this is invoked by ci-gate.yml (the single required check); -# push-to-main and manual runs remain standalone. -on: - workflow_dispatch: - push: - branches: [main] - paths: - - 'integrations/claude-code-plugin/**' - - '.github/workflows/claude-code-plugin-checks.yml' - workflow_call: - -jobs: - test: - runs-on: ubuntu-latest - strategy: - fail-fast: false - matrix: - python-version: ["3.10", "3.11", "3.12"] - steps: - - uses: actions/checkout@v4 - - - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v5 - with: - python-version: ${{ matrix.python-version }} - - - name: Install test tooling - run: pip install pytest ruff - # The plugin itself has zero runtime dependencies — nothing else to install. - - - name: Check manifests are valid JSON - working-directory: integrations/claude-code-plugin - run: | - for f in .claude-plugin/plugin.json .mcp.json hooks/hooks.json; do - jq empty "$f" || (echo "Invalid JSON: $f" && exit 1) - done - - - name: Lint - working-directory: integrations/claude-code-plugin - run: python3 -m ruff check . - - - name: Run tests - working-directory: integrations/claude-code-plugin - run: python3 -m pytest tests -q diff --git a/.github/workflows/deepseek-plugin-checks.yml b/.github/workflows/deepseek-plugin-checks.yml index ecaeb6da6..6054f4832 100644 --- a/.github/workflows/deepseek-plugin-checks.yml +++ b/.github/workflows/deepseek-plugin-checks.yml @@ -8,6 +8,7 @@ on: branches: [main] paths: - 'integrations/deepseek-plugin/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/deepseek-plugin-checks.yml' workflow_call: @@ -84,7 +85,5 @@ jobs: - name: Build run: cd integrations/deepseek-plugin && pnpm build - - name: Verify dist output exists - run: | - test -f integrations/deepseek-plugin/dist/index.js || (echo "Build output missing: dist/index.js" && exit 1) - test -f integrations/deepseek-plugin/dist/index.d.ts || (echo "Build output missing: dist/index.d.ts" && exit 1) + - name: Verify package artifact + run: python3 integrations/agent-plugin-core/conformance/artifacts.py deepseek diff --git a/.github/workflows/mem0-plugin-checks.yml b/.github/workflows/mem0-plugin-checks.yml deleted file mode 100644 index 87eda7ada..000000000 --- a/.github/workflows/mem0-plugin-checks.yml +++ /dev/null @@ -1,58 +0,0 @@ -name: Mem0 Plugin Checks - -# On PRs this is invoked by ci-gate.yml (the single required check); -# push-to-main and manual runs remain standalone. -# -# Covers the Python plugin (scripts/ + tests/). The nested .opencode-plugin/ -# is a separate package with its own workflow (opencode-plugin-checks.yml). -on: - workflow_dispatch: - push: - branches: [main] - paths: - - 'integrations/mem0-plugin/**' - - '!integrations/mem0-plugin/.opencode-plugin/**' - - '.github/workflows/mem0-plugin-checks.yml' - workflow_call: - -jobs: - test: - runs-on: ubuntu-latest - strategy: - fail-fast: false - matrix: - python-version: ["3.10", "3.11", "3.12"] - steps: - - uses: actions/checkout@v4 - - - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v5 - with: - python-version: ${{ matrix.python-version }} - - - name: Install dependencies - working-directory: integrations/mem0-plugin - run: | - pip install -r requirements.txt - pip install pytest - - - name: Verify hook entry points are executable - working-directory: integrations/mem0-plugin - run: | - missing=$(find scripts -name '*.sh' ! -name '_*' ! -perm -u+x -print) - if [ -n "$missing" ]; then - echo "Hook entry points must be executable:" - echo "$missing" - exit 1 - fi - - - name: Check hook manifests are valid JSON - working-directory: integrations/mem0-plugin - run: | - for f in plugin.json mcp_config.json hooks.json hooks/*.json; do - jq empty "$f" || (echo "Invalid JSON: $f" && exit 1) - done - - - name: Run tests - working-directory: integrations/mem0-plugin - run: pytest -q diff --git a/.github/workflows/openclaw-checks.yml b/.github/workflows/openclaw-checks.yml index ec3bfee54..be6651f60 100644 --- a/.github/workflows/openclaw-checks.yml +++ b/.github/workflows/openclaw-checks.yml @@ -8,6 +8,7 @@ on: branches: [main] paths: - 'integrations/openclaw/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/openclaw-checks.yml' workflow_call: @@ -93,7 +94,5 @@ jobs: - name: Build run: cd integrations/openclaw && pnpm build - - name: Verify dist output exists - run: | - test -f integrations/openclaw/dist/index.js || (echo "Build output missing: dist/index.js" && exit 1) - test -f integrations/openclaw/dist/index.d.ts || (echo "Build output missing: dist/index.d.ts" && exit 1) + - name: Verify package artifact + run: python3 integrations/agent-plugin-core/conformance/artifacts.py openclaw diff --git a/.github/workflows/opencode-plugin-cd.yml b/.github/workflows/opencode-plugin-cd.yml index bf37f739f..6c2e11564 100644 --- a/.github/workflows/opencode-plugin-cd.yml +++ b/.github/workflows/opencode-plugin-cd.yml @@ -25,7 +25,7 @@ jobs: id-token: write defaults: run: - working-directory: integrations/mem0-plugin/.opencode-plugin + working-directory: integrations/opencode-plugin steps: - uses: actions/checkout@v4 with: diff --git a/.github/workflows/opencode-plugin-checks.yml b/.github/workflows/opencode-plugin-checks.yml index f1d9585f1..4fb95fc6c 100644 --- a/.github/workflows/opencode-plugin-checks.yml +++ b/.github/workflows/opencode-plugin-checks.yml @@ -7,7 +7,8 @@ on: push: branches: [main] paths: - - 'integrations/mem0-plugin/.opencode-plugin/**' + - 'integrations/opencode-plugin/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/opencode-plugin-checks.yml' workflow_call: @@ -16,7 +17,7 @@ jobs: runs-on: ubuntu-latest defaults: run: - working-directory: integrations/mem0-plugin/.opencode-plugin + working-directory: integrations/opencode-plugin steps: - uses: actions/checkout@v4 @@ -34,6 +35,6 @@ jobs: - name: Build run: bun run build - - name: Verify dist output exists - run: | - test -f dist/index.js || (echo "Build output missing: dist/index.js" && exit 1) + - name: Verify package artifact + working-directory: . + run: python3 integrations/agent-plugin-core/conformance/artifacts.py opencode diff --git a/.github/workflows/pi-agent-plugin-checks.yml b/.github/workflows/pi-agent-plugin-checks.yml index 258863f9e..7e4dc5a0c 100644 --- a/.github/workflows/pi-agent-plugin-checks.yml +++ b/.github/workflows/pi-agent-plugin-checks.yml @@ -8,6 +8,7 @@ on: branches: [main] paths: - 'integrations/pi-agent-plugin/**' + - 'integrations/agent-plugin-core/typescript/**' - '.github/workflows/pi-agent-plugin-checks.yml' workflow_call: @@ -84,9 +85,5 @@ jobs: - name: Build run: cd integrations/pi-agent-plugin && pnpm build - - name: Verify dist output exists - run: | - test -f integrations/pi-agent-plugin/dist/index.js || (echo "Build output missing: dist/index.js" && exit 1) - test -f integrations/pi-agent-plugin/dist/index.d.ts || (echo "Build output missing: dist/index.d.ts" && exit 1) - test -f integrations/pi-agent-plugin/dist/entry.js || (echo "Build output missing: dist/entry.js" && exit 1) - test -f integrations/pi-agent-plugin/dist/entry.d.ts || (echo "Build output missing: dist/entry.d.ts" && exit 1) + - name: Verify package artifact + run: python3 integrations/agent-plugin-core/conformance/artifacts.py pi-agent diff --git a/.gitignore b/.gitignore index 01ee55a0c..c950ac598 100644 --- a/.gitignore +++ b/.gitignore @@ -14,6 +14,10 @@ server/.env # Distribution / packaging .Python build/ +!integrations/agent-plugin-core/build/ +!integrations/agent-plugin-core/build/*.py +!integrations/agent-plugin-core/build/schemas/ +!integrations/agent-plugin-core/build/schemas/*.json develop-eggs/ dist/ downloads/ diff --git a/.kimi-plugin/marketplace.json b/.kimi-plugin/marketplace.json index faaed4998..8ff2134b4 100644 --- a/.kimi-plugin/marketplace.json +++ b/.kimi-plugin/marketplace.json @@ -5,11 +5,11 @@ { "id": "mem0", "displayName": "Mem0", - "version": "0.1.0", - "description": "Persistent memory for Kimi Code. Remembers decisions, patterns, and preferences across sessions.", + "version": "0.3.1", + "description": "Cross-session memory and token savings for coding agents.", "homepage": "https://mem0.ai", "keywords": ["memory", "personalization", "mcp", "semantic-search"], - "source": "https://github.com/mem0ai/mem0/tree/main/integrations/mem0-plugin" + "source": "https://github.com/mem0ai/mem0/tree/main/integrations/kimi-plugin" } ] } diff --git a/AGENTS.md b/AGENTS.md index d91b8ee7b..ba1b72f47 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,7 +14,7 @@ This is a polyglot monorepo and **every package sets its own rules**. Read the ` - Modify anything in `.github/workflows/` without explicit maintainer approval. Publishing credentials are pinned to workflow filenames. - Commit `.env` files, API keys, or credentials. - Skip pre-commit hooks. -- Use npm or yarn in TypeScript packages. This repo is pnpm-only (Bun in `.opencode-plugin/`). +- Use npm or yarn in TypeScript packages. This repo is pnpm-only (Bun in `integrations/opencode-plugin/`). - Use `require()` in TypeScript. ES module `import` syntax only. - Mix up linter configs. Root Python is ruff at line length **120**, `cli/python/` is ruff at **100**, `cli/node/` is Biome, `mem0-ts/` is Prettier, `integrations/vercel-ai-sdk/` is ESLint. - Add Python dependencies to the core `dependencies` list in `pyproject.toml`. Use an optional group. diff --git a/docs/changelog/sdk.mdx b/docs/changelog/sdk.mdx index db9d06396..454f86964 100644 --- a/docs/changelog/sdk.mdx +++ b/docs/changelog/sdk.mdx @@ -2060,6 +2060,29 @@ A full-featured command-line interface for Mem0, available in both Python and No + + +**Changed:** +- Consolidated the coding-agent integrations into `integrations/agent-plugin-core/`: one Python runtime, one TypeScript utility library, and six canonical Python-plugin skill templates. Native adapters retain each host's event contracts and capabilities. +- Python plugins ship generated, self-contained `core/` and `skills/` directories. Builds validate portable schemas and skills, parse native JSON, and reject generated-file drift, missing files, stale generated files, and symlinks. TypeScript packages bundle the shared source into their distributable JavaScript and verify their entry points. +- Replaced the old `integrations/mem0-plugin/` layout with native host directories and one portable `integrations/mem0-agent-plugin/` package. Updated marketplace paths, installation guides, and integration-skill links. OpenCode now lives in `integrations/opencode-plugin/`. +- Native Python plugins expose one local, read-only `search_memories` MCP tool and six skills: search, remember, forget, status, pause, and resume. The shared search tool no longer accepts `run_id`; session IDs remain internal metadata. This local tool is separate from the hosted Mem0 MCP server's tool set. + +**Fixes:** +- Hooks, controls, MCP servers, and detached workers use the same host-specific data directory. Detached workers retain the host identity and telemetry source; `--plugin-data-dir` reaches the shared resolver. +- Session-end workers flush the conversation already captured by hooks. Repeated and concurrent response hooks no longer duplicate an answer, while identical answers after separate prompts are preserved. +- Shared prompts and responses are redacted without the previous 6,000-character cutoff. Python extraction splits oversized messages without dropping text to enforce each request's input budget. Flush event selection and claims share one write transaction, delayed handoffs are replaced atomically, and permanent HTTP polling errors fail promptly. +- Extraction instructions refer to the current coding agent. Python redaction covers JSON-shaped credentials; both telemetry runtimes recursively remove sensitive keys, including keys inside nested lists. +- New Git repository writes use a hash of the remote identity in `agent_id`. Search and explicit shared-memory deletion include both current and legacy repository IDs within the repository's `app_id`. Existing memories are not rewritten. Legacy IDs retain their original ambiguity for matching owner/repository names on different Git hosts. + +**Packaging:** +- Claude Code, Cursor, Codex, Kimi, Antigravity, and the portable Python bundle are versioned at `0.3.1`. OpenCode, Pi Agent, and DeepSeek Harness are `0.3.0`; OpenClaw is `1.1.0`. Each host's changes and upgrade considerations are listed in its tab. +- Python and TypeScript CI run their respective runtime suites. Package checks build the installable artifacts, check generated-file consistency, and reject TypeScript output that still imports monorepo source. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + **Fixes:** @@ -2346,8 +2369,100 @@ Initial release of the Mem0 plugin for Claude Code and Cursor, followed by Codex + + + + +**Changed:** +- Extracted hook orchestration and memory behavior into the shared Python core; Claude transcript parsing remains in its native adapter. The installed package contains the generated runtime rather than importing files outside its plugin directory. +- Preserves the public `mem0` name, hook declarations, MCP launch configuration, user configuration, and `mem0:sidekick` worktree behavior. The manifest and marketplace version advance from `0.3.0` to `0.3.1` so installations can identify the update. + +**Fixes:** +- Session-end extraction no longer appends a final answer already captured from the transcript while an earlier extraction was running. +- Existing repository memories remain searchable after the shared-ID change; explicit shared-memory deletion also covers the legacy ID. Background workers and control skills consistently use Claude's data directory. +- Receives the shared JSON-secret redaction and nested telemetry filtering fixes. The local search tool exposes query, result count, category, and scope without asking the agent for a session ID. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + + + + + + + +**Changed:** +- Moves from the legacy shared editor-plugin directory to a native `integrations/cursor-plugin/` package with generated Python core and skills, Cursor variables, local MCP configuration, and a native Sidekick. +- Translates Cursor conversation/workspace fields, response and summary fields, tool outcomes, and subagent lifecycle events into the shared runtime. The Sidekick searches memory itself because Cursor's subagent-start response cannot inject parent context. + +**Fixes:** +- Adapter errors are logged and exit successfully so a memory failure does not terminate the host hook. Sidekick telemetry reports zero parent-injected context because this host uses self-search. +- Repeated response events and Stop/session-end capture do not duplicate the same answer. +- Receives shared data-directory handling, background-worker identity, redaction, and legacy-memory retrieval fixes. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + + + + + + + +**Changed:** +- Moves from the legacy shared editor-plugin directory to a native `integrations/codex-plugin/` package, with generated Python core and six skills, a local search MCP server, and native lifecycle hooks. +- Resolves MCP repository searches from Codex workspace metadata when supplied. Control skills, hooks, and workers use the same plugin data directory. +- Native subagent start/stop hooks supply parent-retrieved memory context and record completions for every native subagent. Named custom agents remain project/user configuration; the plugin does not distribute a named Codex Sidekick. + +**Fixes:** +- Receives shared credential redaction, legacy-memory retrieval, background-worker identity, and duplicate-response fixes. Codex tool outcomes use available structured failure indicators; missing outcome information is recorded as unknown. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + + + + + + + +**Added:** +- One portable package at `integrations/mem0-agent-plugin/`, using the Agent Plugins 1.0.0 root `plugin.json`, `mcp.json`, and fixed `skills/` locations. +- Ships a local, read-only `search_memories` server and the six shared memory skills. Uses `PLUGIN_ROOT` for bundled files and `PLUGIN_DATA` for persistent plugin state; all package files remain inside the installable directory. + +**Packaging:** +- Generated from the shared Python runtime and skill templates. Builds validate the manifest, MCP configuration, skills, and generated-file consistency. +- Host lifecycle hooks and native Sidekick declarations remain in the native plugin packages; the portable package does not claim automatic lifecycle capture or host-specific subagent isolation. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + + + + + +**Changed:** +- Moved the source from `integrations/mem0-plugin/.opencode-plugin/` to `integrations/opencode-plugin/`, retaining the `@mem0/opencode-plugin` package name and native OpenCode hooks. +- Reuses shared conversation preparation, redaction, scoping, and telemetry. Builds a self-contained Bun/ESM `dist/index.js` and publishes its TypeScript entry declaration. +- Global memory tool scope requires the user to enable it in plugin settings first; empty and wildcard identities are rejected. +- Retains the seven commands for context loading, search, remember, forget, scope, status, and tour; bundled skills continue loading through OpenCode's native configuration. + +**Removed:** +- Removed auto-Dream consolidation, its gates and state handling, and the Dream and pin skills/commands. Existing configurations and workflows that use these features must be updated. + +**Builds:** +- Updated build and publish paths for the relocated source directory; the existing release tag prefix and publishing workflow filename are unchanged. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + **Fixes:** @@ -2428,6 +2543,22 @@ Initial release of the Mem0 plugin for Claude Code and Cursor, followed by Codex + + +**Changed:** +- Ships a native package with generated Python core and six skills, a local search MCP server, a Sidekick declaration, and an adapter for `PreInvocation`, `PostToolUse`, and `Stop`. +- Normalizes conversation IDs, transcript paths, workspace paths, tool calls, and errors. The adapter captures all completed user/assistant transcript messages incrementally, including later-turn intent, without replaying earlier messages. +- Sidekick searches Mem0 itself; this plugin does not provide Claude Code's worktree isolation. + +**Fixes and host limitations:** +- Accepts `MEM0_CWD` as an explicit workspace fallback when the host omits `workspacePaths`. Skips capture when neither is available, rather than writing under an unrelated directory. +- Documents global MCP registration with `agy mcp add` when the host does not register the plugin-scoped server. Recall on the initial invocation depends on the host providing a prompt or readable transcript. +- Receives the shared data-directory, redaction, legacy-memory retrieval, and duplicate-response fixes. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + **Fixes:** @@ -2504,6 +2635,24 @@ Existing memories written by the previous versions are not rewritten. If your me + + +**Changed:** +- Ships a self-contained native package with six skills, nine lifecycle hooks, a local search MCP server, and a native Sidekick declaration. Added a dedicated installation and troubleshooting guide. +- Translates Kimi's session, prompt, tool, compaction, shutdown, and subagent events into the shared Python runtime. Sidekicks receive parent memory context through Kimi's native lifecycle. + +**Fixes:** +- Recovers completed assistant output from Kimi's indexed v2 wire transcript when Stop events omit the response text. Repeated Sidekick invocations receive distinct run identifiers. Stops without a host ID are left uncorrelated when multiple matching runs are active, preserving their responses without assigning them to the wrong run. +- Keeps controls, hooks, MCP, and detached workers on the same Kimi data directory and preserves host identity in background workers. +- Receives the shared redaction, legacy-memory retrieval, and duplicate-response fixes. + +**Host compatibility:** +- Documents `CHOKIDAR_USEPOLLING=1` for the observed macOS watcher issue. Filesystem isolation remains Kimi's responsibility. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + **Initial release** of the Mem0 plugin for Kimi Code, sharing its scripts, skills, and marketplace listing with the Claude Code / Cursor / Codex / Antigravity plugin family ([#6919](https://github.com/mem0ai/mem0/pull/6919)) @@ -2520,6 +2669,23 @@ Existing memories written by the previous versions are not rewritten. If your me + + +**Changed:** +- Reuses shared conversation preparation, redaction, and telemetry while retaining OpenClaw's native memory backend, tools, CLI, and Platform/OSS modes. +- Continues to publish a self-contained ESM package under `@mem0/openclaw-mem0`; the plugin manifest and package version now agree. + +**Removed:** +- Removed Dream consolidation: automatic scheduling and locking, `openclaw mem0 dream`, Dream configuration, the memory-dream skill, and Dream-state public artifacts. Triage, recall, and memory/entity artifacts remain available. Update configurations or integrations that use the removed Dream surface. + +**Fixes:** +- `openclaw mem0 status` handles an unconfigured installation without crashing and directs users to setup. +- Telemetry removes sensitive properties recursively and uses the shared failure-safe delivery implementation. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + **Security:** @@ -2790,6 +2956,24 @@ Existing memories written by the previous versions are not rewritten. If your me + + +**Changed:** +- Reuses shared conversation preparation, memory formatting, project/session/global scope utilities, and telemetry while preserving Pi's native extension API and `@mem0/pi-agent-plugin` package name. +- Pi loads the built `dist/entry.js` extension instead of executing source TypeScript from an installed package. Builds also publish the library entry point and declarations. + +**Removed:** +- Removed Dream consolidation and pin commands, skills, configuration, types, and exports. The remaining commands are remember, search, forget, tour, scope, and status. Update integrations that import removed APIs or invoke removed commands. + +**Fixes:** +- Global memory tool scope requires the user to select `/mem0-scope global` or configure a global default first. Empty and wildcard identities are rejected. +- Memory update and delete accept the `mem0:` and `[mem0:]` citations displayed in tool results, as well as raw IDs. +- Shared capture preparation filters conversation roles and redacts content; telemetry removes sensitive keys inside nested structures. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + **Security:** @@ -2885,6 +3069,23 @@ Existing memories written by the previous versions are not rewritten. If your me + + +**Added:** +- Automatic recall during `system-prompt/assemble`, using the latest human prompt and avoiding repeated context injection within a session. +- Automatic capture from the durable `session/event` stream after a completed turn. Interrupted or incomplete turns are not sent through this automatic capture path. `autoRecall` and `autoCapture` default to `true` and can be disabled. + +**Changed:** +- Reuses shared lifecycle, redaction, identity, and telemetry utilities while retaining the explicit `search_memory` and `add_memory` tools and their per-call agent/session scope. Cross-user `userId` overrides now require operator opt-in with `allowUserOverride: true`. +- Publishes a self-contained ESM artifact under `@mem0/deepseek-plugin`; native Harness services and the Mem0 SDK remain external dependencies. Plugin cleanup remains tied to the native Cordis lifecycle. + +**Host compatibility:** +- Supports the declared Harness runtime dependencies and documents the macOS watcher workaround. The plugin does not bundle a named Sidekick or provide child filesystem isolation. + +[#7203](https://github.com/mem0ai/mem0/pull/7203) + + + **New Features:** diff --git a/docs/docs.json b/docs/docs.json index 269cf9426..c1c4887d2 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -373,6 +373,7 @@ "integrations/claude-ai", "integrations/cursor", "integrations/codex", + "integrations/kimi", "integrations/opencode", "integrations/antigravity" ] diff --git a/docs/integrations/antigravity.mdx b/docs/integrations/antigravity.mdx index a3017d2ef..d9229bf2f 100644 --- a/docs/integrations/antigravity.mdx +++ b/docs/integrations/antigravity.mdx @@ -1,16 +1,20 @@ --- title: Antigravity -description: "Add persistent memory to Google Antigravity with the Mem0 plugin: MCP server, lifecycle hooks, and slash commands." +description: "Add persistent memory to Google Antigravity with the Mem0 plugin: a search tool, lifecycle hooks, and memory skills." --- -Add persistent memory to [**Google Antigravity**](https://antigravity.google) (`agy` CLI and Desktop IDE) with the Mem0 plugin. Your agent forgets everything between sessions. Mem0 fixes that by storing decisions, preferences, and learnings so they carry over automatically. +Add persistent memory to [**Google Antigravity**](https://antigravity.google) (`agy` CLI and Desktop IDE) with the Mem0 plugin. The plugin captures completed work, and Antigravity can search those memories in later sessions. + +Current plugin version: `0.3.1`. ## Prerequisites 1. A Mem0 API key (starts with `m0-`): - Get your API key (free sign-up at app.mem0.ai) -2. Add it to your shell profile so it persists across sessions: +2. Python 3.10 or newer available as `python3` + +3. Add the key to your shell profile so it persists across sessions: ```bash zsh @@ -26,22 +30,33 @@ echo 'export MEM0_API_KEY="m0-your-api-key"' >> ~/.bashrc && source ~/.bashrc **Option A: degit** (recommended): + + When upgrading an existing installation, move `~/.gemini/config/plugins/mem0` aside first. Installing over that directory can leave obsolete files from older plugin versions. + + ```bash # Install the plugin (MCP server, hooks, scripts) -npx degit mem0ai/mem0/integrations/mem0-plugin ~/.gemini/config/plugins/mem0 +npx degit mem0ai/mem0/integrations/antigravity-plugin ~/.gemini/config/plugins/mem0 ``` -This installs the MCP server, lifecycle hooks, and shared scripts. +This installs the generated Antigravity bundle with the Mem0 server, lifecycle hooks, and six memory skills. ## What's Included | Component | Included | |-----------|:--------:| -| MCP Server (9 memory tools) | Yes | +| Local MCP Server (`search_memories`) | Yes | | Lifecycle Hooks | Yes | -| 16 Slash Commands | Yes | +| 6 Memory Skills | Yes | +| Sidekick Agent | Yes | -## Available MCP Tools +## Memory tools and skills + +The full plugin exposes a focused `search_memories` tool and captures completed work through lifecycle hooks. Six skills provide search, status, remember, forget, pause, and resume workflows. A native sidekick can handle a bounded task in the workspace assigned by Antigravity; it does not claim Claude Code worktree isolation. + +If you need the full set of direct CRUD tools, connect the [hosted Mem0 MCP server](/platform/mem0-mcp) separately instead of installing both configurations under the same server name. + +## Hosted MCP tools | Tool | Description | |------|-------------| @@ -57,23 +72,46 @@ This installs the MCP server, lifecycle hooks, and shared scripts. ## Lifecycle Hooks -The plugin uses the same shell scripts as Claude Code, Cursor, and Codex: hooks bridge environment variables using `${extensionPath}` (Antigravity's plugin-root token). +The plugin translates Antigravity events into the shared Mem0 capture lifecycle. | Hook | Event | What it does | |------|-------|-------------| -| **Session start** | `SessionStart` | Loads prior memories and displays status banner | -| **User prompt** | `UserPromptSubmit` | Searches relevant memories before each message | -| **Pre-tool** | `PreToolUse` | Blocks MEMORY.md writes, enforces `user_id`/`app_id` on mem0 tools | -| **Post-tool** | `PostToolUse` | Tracks stats, scans bash errors for related memories | -| **Stop** | `Stop` | Stores a session summary at the end of every assistant turn (not just at session end) | +| **Invocation** | `PreInvocation` | Initializes the first invocation and recovers pending capture | +| **Post-tool** | `PostToolUse` | Records useful tool results and failures | +| **Stop** | `Stop` | Captures the completed exchange for a later session | -What you type is stored as yours. What the agent produces (session summaries and compaction summaries) is stored as the assistant's, so its suggestions never become your stated preferences. +Recall is explicit through `search_memories` and the search skill. The current Antigravity adapter does not inject query-specific memory during `PreInvocation` because that event does not include the user's prompt. + +What you type is stored as yours. Captured agent summaries are stored as the assistant's, so its suggestions never become your stated preferences. + +## Antigravity 1.1.25 workarounds + +Antigravity 1.1.25 can validate a plugin's `mcp_config.json` without loading the server. Check whether Mem0 is registered: + +```bash +agy mcp list +``` + +If `mem0` is absent, register the plugin's bundled server with Antigravity's supported global MCP command, then start a new session: + +```bash +agy mcp add mem0 python3 ~/.gemini/config/plugins/mem0/core/mcp_server.py +``` + +The same release can send hooks with an empty `workspacePaths` array. For the `agy` CLI, pass the current repository explicitly: + +```bash +MEM0_CWD="$PWD" agy +``` + +The adapter uses `MEM0_CWD` only when Antigravity omits the workspace. If neither value is available, it skips recall and capture instead of writing memories under an incorrect repository scope. ## Troubleshooting - **No tools appearing**: Restart your Antigravity session after installation -- **"Connection failed"**: Verify your key is set: `echo $MEM0_API_KEY` -- **MCP 401 Unauthorized**: If `${MEM0_API_KEY}` interpolation doesn't work in your `agy` version, replace with your literal key in `mcp_config.json` +- **No `mem0` entry in `agy mcp list`**: Use the Antigravity 1.1.25 global MCP registration command above +- **Hooks do not capture in the CLI**: Launch `agy` with `MEM0_CWD="$PWD"` +- **"Connection failed"**: Verify your key is set without printing it: `test -n "$MEM0_API_KEY" && echo configured` diff --git a/docs/integrations/claude-code.mdx b/docs/integrations/claude-code.mdx index 0dd0b153f..e185758dc 100644 --- a/docs/integrations/claude-code.mdx +++ b/docs/integrations/claude-code.mdx @@ -5,6 +5,8 @@ description: "Persistent cross-session memory for Claude Code. Install once, mem Claude Code forgets everything between sessions. This plugin fixes that. Install it, work normally, and Claude remembers what happened across sessions. +Current plugin version: `0.3.1`. + ## Prerequisites 1. A Mem0 Platform account and API key (starts with `m0-`): @@ -71,6 +73,8 @@ Review its result and send any corrections back to the same sidekick. Changes stay in the sidekick's worktree until the main agent reviews and copies them over. By default the worktree branches from the repo's default branch. Set `worktree.baseRef` to `"head"` in your Claude settings to branch from the current commit instead. Uncommitted changes are not copied into the sidekick's worktree. +At startup, Sidekick receives the memories already recalled for its parent session. It can also call the Mem0 search tool for its assigned task. Start and completion hooks track its work locally; completing a Sidekick task does not independently send a memory-extraction request. Sidekick returns its result, validation, and a local commit when it changes files, so the main agent can review the work before incorporating it. + ## How it works The plugin follows a simple cycle: capture during a session, extract memories in the background, recall in the next session. @@ -81,11 +85,11 @@ The plugin follows a simple cycle: capture during a session, extract memories in **Step by step:** -1. **Capture.** Hooks save the main agent's activity locally: user messages, Claude's answers, changed files, and short test/build results. Subagent (sidekick) output is excluded. No model calls, no blocking. +1. **Capture.** Hooks save the main agent's activity locally: user messages, Claude's answers, changed files, and short test/build results. Direct Sidekick lifecycle records stay local. Subagent results included in the main transcript can provide supporting evidence for extraction; the main agent's final response establishes the outcome. Capture does not call a model. 2. **Flush.** After every five completed exchanges, a detached background worker sends that batch to Mem0. Large exchanges flush sooner. Ending or compacting the session flushes anything remaining. If the session sits idle, an auto-flush runs after five minutes (configurable with `MEM0_CODE_IDLE_FLUSH_SECONDS`). The timer resets on each new exchange. The worker survives Claude Code exiting. -3. **Extract.** Each flush sends a single `add` call with `agent_id` (the project identity), `user_id` (you), `app_id` (the repository), and `run_id` (the session). Mem0 classifies each extracted memory as shared project knowledge or a personal preference. +3. **Extract.** Each flush sends `add` calls with `agent_id` (the project identity), `user_id` (you), `app_id` (the repository), and `run_id` (the session). Prompts and responses are redacted without a character cutoff. Large conversations are split across requests without dropping message text. Mem0 classifies each extracted memory as shared project knowledge or a personal preference. 4. **Recall.** On the next session's first prompt, the plugin searches automatically and supplies up to five relevant memories. No model is called to write the query. @@ -110,10 +114,14 @@ Every memory carries identifiers showing where it came from: | Identifier | What it is | Example | | --- | --- | --- | | `user_id` | You (personal memory only) | Your Mem0 user ID | -| `agent_id` | The project identity (shared memory only) | `acme-payments-api` | +| `agent_id` | The project identity (shared memory only) | `acme-payments-api-` | | `app_id` | The repository (scopes both lanes) | `acme-payments-api` | | `run_id` | The Claude Code session | The session ID | +New Git repository memories use an `agent_id` with a hash of the Git remote identity so matching owner/repository names on different hosts stay separate. Searches also include the previous unhashed `agent_id`, scoped by the repository's `app_id`, so existing shared memories remain available after upgrading. Those older memories retain their original namespace, which did not distinguish Git hosts. Local folders continue using a hash of their path. + +Explicitly forgetting shared project memory with `--include-project-memory` covers both repository IDs. Without that option, shared memories are preserved. + A search returns the union of shared project memory and your personal preferences. The scope narrows the project part: | Scope | What you get | @@ -124,7 +132,7 @@ A search returns the union of shared project memory and your personal preference The `dir` scope is hierarchical: a parent directory sees everything in its children, but a child never sees the parent's memories. -Pass `--run-id ` to see only what one specific session recorded. Set the default scope with the `search_scope` setting or `MEM0_CODE_SEARCH_SCOPE` env var. +The search tool accepts an optional `run_id` with every scope (`repo`, `dir`, and `mine`); `/mem0:search` exposes it as `--run-id session-id`. It restricts both shared and personal results to memories saved in that coding-agent session. Omit it to search across sessions. This is a memory filter, not a label for the session making the request; automatically filtering by the current session would hide earlier-session memories. Each memory update still records the session's `run_id`. Set the default scope with the `search_scope` setting or `MEM0_CODE_SEARCH_SCOPE` env var. ## Settings diff --git a/docs/integrations/codex.mdx b/docs/integrations/codex.mdx index 7b2be4b55..33f09b16c 100644 --- a/docs/integrations/codex.mdx +++ b/docs/integrations/codex.mdx @@ -1,9 +1,11 @@ --- title: Codex -description: "Add persistent memory to OpenAI Codex with the Mem0 plugin: MCP server, lifecycle hooks, and SDK skill." +description: "Add persistent memory to OpenAI Codex with automatic capture, automatic recall, a search tool, and six memory skills." --- -Add persistent memory to [**OpenAI Codex**](https://openai.com/codex/) with the Mem0 plugin. Codex forgets everything between tasks. This plugin fixes that by connecting to Mem0's cloud memory layer via MCP, automatically capturing learnings at key lifecycle points, and retrieving relevant context before every response. +Add persistent memory to [**OpenAI Codex**](https://openai.com/index/codex/) with the Mem0 plugin. Codex forgets everything between tasks. This plugin fixes that by connecting to Mem0's cloud memory layer via MCP, automatically capturing learnings at key lifecycle points, and retrieving relevant context before every response. + +Current plugin version: `0.3.1`. ## Prerequisites @@ -15,7 +17,9 @@ Before setting up Mem0 with Codex, ensure you have: 2. OpenAI Codex access -3. Your API key added to your shell profile (persists across sessions): +3. Python 3.10 or newer available as `python3` + +4. Your API key added to your shell profile (persists across sessions): ```bash zsh @@ -33,7 +37,7 @@ source ~/.bashrc ### Option A: Plugin Marketplace (Recommended) -Install the full plugin including MCP server, lifecycle hooks, and SDK skill. +Install the full plugin, including automatic capture and recall, the `search_memories` tool, and six memory skills. 1. Add the Mem0 marketplace: @@ -88,7 +92,7 @@ codex plugin marketplace remove mem0-plugins # unregister the marketplace enti To update, run `codex plugin marketplace upgrade` to pull the latest from the Mem0 repo. - After either option, start a new Codex task and ask: *"List my mem0 entities"* or *"Search my memories for hello"*. If the `mem0` tools appear and respond, you're all set. + After Option A, ask Codex to search memory for a recent project decision. After Option B, ask it to list your Mem0 entities. If the matching tool responds, the connection is ready. ## Codex Cloud @@ -106,13 +110,16 @@ Lifecycle hooks that shell out to local scripts (Option A) are not applicable in | Component | Plugin Install | MCP Only | |-----------|:--------------:|:--------:| -| MCP Server (9 memory tools) | Yes | Yes | -| Lifecycle Hooks | Opt-in (see below) | No | -| Mem0 SDK Skill | Yes | No | +| Memory tools | `search_memories` | 9 CRUD tools | +| Lifecycle hooks | Yes | No | +| Native subagent memory lifecycle | Yes | No | +| Memory skills | 6 | No | -## Available MCP Tools +Codex plugins cannot currently bundle a named custom agent. If you define project or user agents under `.codex/agents/` or `~/.codex/agents/`, the full Mem0 plugin gives every native subagent the parent turn's retrieved memory context and records its completed result. -Once installed, the following tools are available in every Codex session: +## Direct MCP tools + +Option B exposes the following hosted tools. The full plugin in Option A uses the focused `search_memories` tool and captures writes through lifecycle hooks. | Tool | Description | |------|-------------| @@ -126,33 +133,20 @@ Once installed, the following tools are available in every Codex session: | `delete_entities` | Delete a user/agent/app/run entity and its memories | | `list_entities` | List users/agents/apps/runs stored in Mem0 | -## Lifecycle Hooks +## Lifecycle hooks -Unlike Claude Code, Codex has no plugin-host mechanism for auto-wiring hooks from an installed plugin: it only reads hooks from `~/.codex/hooks.json` (or `/.codex/hooks.json`). Installing the plugin (Option A) does **not** turn hooks on by itself. To enable them, run the bundled installer once against your local clone: - -```bash -python3 /integrations/mem0-plugin/scripts/install_codex_hooks.py -``` - -This merges Mem0's entries into `~/.codex/hooks.json` and is idempotent (safe to re-run after upgrading). It also requires the `codex_hooks` feature flag in `~/.codex/config.toml`: - -```toml -[features] -codex_hooks = true -``` - -The installer prints a reminder if the flag isn't set. Restart Codex after installing hooks or editing the config. To remove: `python3 .../install_codex_hooks.py --uninstall`. - -Once enabled, Mem0 hooks into Codex's lifecycle to automatically manage memory: +Option A registers the hooks with the plugin. No separate hook installer or global `hooks.json` edit is required. | Hook | Event | What it does | |------|-------|-------------| | **Session start** | `SessionStart` | Loads prior memories and displays status banner | | **User prompt** | `UserPromptSubmit` | Searches relevant memories before each message | -| **Pre-tool (3 handlers)** | `PreToolUse` | Blocks MEMORY.md writes; enforces `user_id`/`app_id` on mem0 tool calls; scans files being read for relevant memory context | -| **Post-tool** | `PostToolUse` | Tracks stats, scans bash errors for related memories | -| **Stop** | `Stop` | Stores a session summary at the end of every assistant turn (not just at session end) | -| **Pre-compact** | `PreCompact` | Stores a summary before the context is compacted | +| **Post-tool** | `PostToolUse` | Records useful tool outcomes for the completed exchange | +| **Subagent start** | `SubagentStart` | Reuses the parent turn's retrieved memory context in the child | +| **Subagent stop** | `SubagentStop` | Records the child transcript path and completed result | +| **Stop** | `Stop` | Captures completed work and starts a background flush when needed | +| **Pre-compact** | `PreCompact` | Flushes pending capture before context compaction | +| **Session end** | `SessionEnd` | Flushes any remaining capture in the background | What you type is stored as yours. What Codex produces (session summaries and compaction summaries) is stored as the assistant's, so its suggestions never become your stated preferences. @@ -181,7 +175,7 @@ You: Add WebSocket support for real-time notification delivery. - **"Connection failed"**: Verify `MEM0_API_KEY` is set: `echo $MEM0_API_KEY` - **No tools appearing**: Restart your Codex session after installation - **Duplicate `mem0` MCP / "tool collision" errors**: You combined Option A with Option B. Remove the `[mcp_servers.mem0]` block from `~/.codex/config.toml`; the plugin registers it automatically -- **Hooks not firing**: Hooks are opt-in and are not installed by the marketplace install itself. Run `scripts/install_codex_hooks.py` (see [Lifecycle Hooks](#lifecycle-hooks)), confirm `codex_hooks = true` is set under `[features]` in `~/.codex/config.toml`, and restart Codex. MCP-only installs (Option B) never include hooks +- **Hooks not firing**: Confirm you installed Option A and restart Codex. MCP-only installs (Option B) do not include hooks. diff --git a/docs/integrations/cursor.mdx b/docs/integrations/cursor.mdx index 03486853b..df54d27dc 100644 --- a/docs/integrations/cursor.mdx +++ b/docs/integrations/cursor.mdx @@ -1,9 +1,11 @@ --- title: Cursor -description: "Add persistent memory to Cursor with the Mem0 MCP server for context-aware coding." +description: "Add persistent memory to Cursor with automatic capture, explicit recall, a search tool, and six memory skills." --- -Add persistent memory to [**Cursor**](https://cursor.com) with the Mem0 MCP server. Your AI assistant forgets everything between sessions. Mem0 fixes that by connecting Cursor to Mem0's cloud memory layer via MCP so you can save and retrieve relevant context during coding sessions. +Add persistent memory to [**Cursor**](https://cursor.com) with the Mem0 plugin. Cursor captures completed work in the background, and its agent can search relevant project context in later sessions. You can also connect only the hosted MCP server when you do not need lifecycle capture. + +Current plugin version: `0.3.1`. ## Prerequisites @@ -15,7 +17,9 @@ Before setting up Mem0 with Cursor, ensure you have: 2. Cursor installed ([cursor.com](https://cursor.com)) -3. Your API key added to your shell profile (persists across sessions): +3. Python 3.10 or newer available as `python3` + +4. Your API key added to your shell profile (persists across sessions): ```bash zsh @@ -35,13 +39,25 @@ source ~/.bashrc ## Installation -### Option A: One-Click Deeplink (MCP Only) +### Option A: Full Plugin (Recommended) + +Add the Mem0 marketplace: + +```bash +cursor-agent plugin marketplace add https://github.com/mem0ai/mem0 +``` + +Open Cursor's **Customize** page or run `/plugins`, select the **Mem0 Plugins** marketplace, and install **Mem0** at user or project scope. Enter your Mem0 API key when Cursor asks for the plugin configuration, then restart Cursor. + +The full plugin includes automatic capture, the `search_memories` recall tool, lifecycle hooks, a sidekick agent, and six memory skills. + +### Option B: One-Click Deeplink (MCP Only) The fastest way to get started. Click the link below to install the Mem0 MCP server directly in Cursor: [Install Mem0 MCP in Cursor](cursor://anysphere.cursor-deeplink/mcp/install?name=mem0&config=eyJtY3BTZXJ2ZXJzIjp7Im1lbTAiOnsidXJsIjoiaHR0cHM6Ly9tY3AubWVtMC5haS9tY3AvIiwiaGVhZGVycyI6eyJBdXRob3JpemF0aW9uIjoiVG9rZW4gJHtlbnY6TUVNMF9BUElfS0VZfSJ9fX19) -### Option B: npx (MCP Only) +### Option C: npx (MCP Only) ```bash npx mcp-add \ @@ -51,7 +67,7 @@ npx mcp-add \ --clients "cursor" ``` -### Option C: Manual Configuration (MCP Only) +### Option D: Manual Configuration (MCP Only) Add the following to your `.cursor/mcp.json`: @@ -73,9 +89,20 @@ Add the following to your `.cursor/mcp.json`: -## Available MCP Tools +## What's included -Once installed, the following tools are available in every Cursor session: +| Component | Full plugin | MCP only | +| --- | :---: | :---: | +| Memory tools | `search_memories` | 9 CRUD tools | +| Automatic capture | Yes | No | +| Recall | `search_memories` | MCP search tools | +| Lifecycle hooks | Yes | No | +| Memory skills | 6 | No | +| Sidekick agent | Yes | No | + +## MCP-only tools + +Options B, C, and D expose the following hosted tools. The full plugin uses the focused `search_memories` tool and captures writes through lifecycle hooks. | Tool | Description | |------|-------------| @@ -89,6 +116,25 @@ Once installed, the following tools are available in every Cursor session: | `delete_entities` | Delete a user/agent/app/run entity and its memories | | `list_entities` | List users/agents/apps/runs stored in Mem0 | +## Lifecycle hooks + +The full plugin translates Cursor's native events into the shared Mem0 memory lifecycle: + +| Cursor event | What Mem0 does | +| --- | --- | +| `sessionStart` | Initializes the project session and recovers pending capture | +| `beforeSubmitPrompt` | Observes the submitted prompt; recall stays tool-driven because this event cannot inject context | +| `postToolUse` / `postToolUseFailure` | Records useful tool results and failures | +| `subagentStart` / `subagentStop` | Records the Mem0 Sidekick lifecycle and completed result | +| `afterAgentResponse` / `stop` | Captures the completed exchange | +| `preCompact` / `sessionEnd` | Flushes pending capture in the background | + + + Cursor's `beforeSubmitPrompt` hook can allow or block a prompt, but it cannot add context to that prompt. The plugin therefore gives the agent `search_memories` and a memory-aware sidekick for recall instead of claiming automatic per-prompt injection. + + +Cursor's `subagentStart` hook also cannot inject parent context. The bundled Sidekick searches Mem0 itself, while the lifecycle hooks correlate its start and completion for later capture. + ## Example Workflow ```text @@ -112,7 +158,7 @@ You: The /orders endpoint is also slow, same pattern as before. ## Troubleshooting - **"Connection failed"**: Verify `MEM0_API_KEY` is set: `echo $MEM0_API_KEY` -- **Duplicate tools**: If you had a previous MCP config for `mem0`, remove it before installing the plugin +- **Duplicate tools**: Do not combine the full plugin with an MCP-only option. Remove the standalone `mem0` MCP entry before installing the plugin. - **No tools appearing**: Go to Cursor Settings > MCP and verify the `mem0` server shows as connected diff --git a/docs/integrations/deepseek-plugin.mdx b/docs/integrations/deepseek-plugin.mdx index f26b08b23..1a9f98a6d 100644 --- a/docs/integrations/deepseek-plugin.mdx +++ b/docs/integrations/deepseek-plugin.mdx @@ -1,16 +1,20 @@ --- title: DeepSeek Harness -description: "Add persistent Mem0 memory to the DeepSeek Harness (Cordis) agent with two native tools: search and add." +description: "Add persistent memory to DeepSeek Harness with automatic recall, automatic capture, and two native Mem0 tools." --- -Add persistent memory to the [**DeepSeek Harness**](https://github.com/deepseek-ai/deepseek-harness) with `@mem0/deepseek-plugin`. The Harness agent forgets everything between sessions. This plugin gives it two Mem0-backed tools so recall and writes persist across runs, sharing the same memory bank you already use from Claude Code, Codex, and other agents. +Add persistent memory to the [**DeepSeek Harness**](https://github.com/deepseek-ai/deepseek-harness) with `@mem0/deepseek-plugin`. The plugin recalls relevant context before a model request, captures completed turns, and provides explicit Mem0 tools when the agent needs them. + +Current package version: `0.3.0`. ## Overview -The plugin registers two agent-callable tools: +The plugin provides automatic memory plus two agent-callable tools: -| Tool | Does | +| Capability | What it does | |---|---| +| Automatic recall | Searches with the latest human prompt and adds unseen results to the model context | +| Automatic capture | Stores the human and assistant messages from each completed turn | | `search_memory` | Recall facts from Mem0 relevant to a query | | `add_memory` | Store a fact in Mem0 for future sessions | @@ -18,7 +22,15 @@ Unlike file-based memory plugins, Mem0 is a managed backend: server-side extract ## How it works -A Cordis plugin is a module exporting `apply(ctx, config)`. This one declares `inject = ['tools']` so it waits for the harness tool registry, then registers the two tools via `ctx.tools.register(...)`. When the plugin unmounts, the tools are removed automatically (Cordis revertible effects). +A Cordis plugin is a module exporting `apply(ctx, config)`. This plugin waits for the Harness tool and system-prompt services, then uses their native extension points: + +- `system-prompt/assemble` recalls memory before a model request. +- `session/event` captures only completed turns from the durable event stream. +- `ctx.tools.register(...)` exposes explicit search and add tools. + +Cordis removes the listeners and tools when the plugin unmounts. Memory failures are fail-open, so a Mem0 outage does not stop the agent from completing its normal work. + +DeepSeek Harness provides subagents through separate host-composition packages. This Mem0 package does not register a named Sidekick or claim child filesystem isolation. A Harness child uses Mem0 only when its own agent preset includes the Mem0 plugin. ## Prerequisites @@ -44,19 +56,28 @@ source ~/.bashrc ## Try it locally -1. Build the plugin: +1. Build and pack the plugin: ```sh cd integrations/deepseek-plugin - pnpm install + pnpm install --frozen-lockfile pnpm build + mkdir -p /tmp/mem0-deepseek-plugin + pnpm pack --pack-destination /tmp/mem0-deepseek-plugin ``` -2. Point the Harness at it. Copy `cordis.example.yml`, set the absolute path to `dist/index.js` and your `userId`, then load it: +2. Install it into a disposable Harness profile so Harness supplies its peer dependencies: ```sh - pnpm dsh web --patch ./integrations/deepseek-plugin/cordis.example.yml + DSH_HOME=/tmp/mem0-dsh-dev pnpm dlx @deepseek-ai/dsh@0.1.1-rc.2 \ + plugin --profile headless add /tmp/mem0-deepseek-plugin/mem0-deepseek-plugin-0.3.0.tgz ``` -3. Open the web UI and ask the agent to remember something, then recall it in a later turn. +3. Copy `cordis.example.yml`, set its installed package path and your `userId`, then load it with the same profile: + ```sh + DSH_HOME=/tmp/mem0-dsh-dev pnpm dlx @deepseek-ai/dsh@0.1.1-rc.2 \ + web --patch ./integrations/deepseek-plugin/cordis.example.yml + ``` + +4. Open the web UI and ask the agent to remember something, then recall it in a later turn. The `cordis.yml` entry looks like this: @@ -65,10 +86,12 @@ The `cordis.yml` entry looks like this: - name: "@deepseek-ai/dsh-tools" - insert: - id: mem0 - name: "/absolute/path/to/integrations/deepseek-plugin/dist/index.js" + name: "/tmp/mem0-dsh-dev/profiles/headless/node_modules/@mem0/deepseek-plugin/dist/index.js" config: # apiKey is read from MEM0_API_KEY when omitted here. userId: "your-user-id" + autoRecall: true + autoCapture: true # host: "https://your-onprem.mem0.ai" # optional: Platform on-prem / dedicated base URL ``` @@ -80,10 +103,25 @@ For a Mem0 Platform on-prem or dedicated deployment, point `config.host` at that |---|---|---|---| | `apiKey` | no | `$MEM0_API_KEY` | Mem0 platform API key | | `userId` | yes | | Default entity that owns the memories | +| `allowUserOverride` | no | `false` | Permit model-selected access to a different user only in a trusted multi-user deployment | | `host` | no | `api.mem0.ai` | Platform base URL (on-prem / dedicated) | +| `autoRecall` | no | `true` | Recall relevant memory before model requests | +| `autoCapture` | no | `true` | Store completed human and assistant turns | Both tools also accept optional per-call `userId`, `agentId`, and `runId` params so a single install can partition memory by entity, agent, or session; when omitted they fall back to the configured `userId`. ## Telemetry -Writes are tagged `source="DEEPSEEK_HARNESS"` so Mem0's backend can attribute usage to this integration. +Writes are tagged `source="DEEPSEEK_HARNESS"` so Mem0 can attribute usage to this integration. Anonymous usage events include operation names, durations, result counts, and coarse failure kinds. Queries, memory text, entity IDs, and API keys are never included. Set `MEM0_TELEMETRY=false` to opt out. + + + This plugin is a developer preview and tracks the evolving DeepSeek Harness plugin API. + + +## Troubleshooting + +- **`MISSING_CREDENTIAL` for `deepseek-official`**: Configure `DEEPSEEK_API_KEY` through Harness's Models page or export it in the shell that launches Harness. +- **`EMFILE: too many open files, watch` on macOS**: Launch Harness with `CHOKIDAR_USEPOLLING=1`. +- **Mem0 tools do not appear**: Run Harness with `--dump-config` and confirm the final composition contains the `mem0` row and the installed `dist/index.js` path. + +Per-call `userId` overrides are rejected unless the operator enables `allowUserOverride: true`. Automatic recall and capture always use the configured user. diff --git a/docs/integrations/kimi.mdx b/docs/integrations/kimi.mdx new file mode 100644 index 000000000..9b66d7603 --- /dev/null +++ b/docs/integrations/kimi.mdx @@ -0,0 +1,109 @@ +--- +title: Kimi Code +description: "Add persistent project memory to Kimi Code with automatic capture, automatic recall, Mem0 skills, and tools." +--- + +Kimi Code forgets project decisions between sessions. The Mem0 plugin captures completed work, recalls relevant context before a response, and gives Kimi explicit memory tools and skills. + +Current plugin version: `0.3.1`. + +## Prerequisites + +1. A Mem0 Platform account and API key: + - [Sign up at app.mem0.ai](https://app.mem0.ai?utm_source=oss&utm_medium=integration-kimi) + - [Get your API key](https://app.mem0.ai/dashboard/api-keys?utm_source=oss&utm_medium=integration-kimi) (starts with `m0-`) + +2. [Kimi Code](https://www.kimi.com/code) with plugin support. + +3. Python 3.10+ and Git on your machine. + +## Quick start + +Export the key in the shell where you start Kimi Code: + +```bash +export MEM0_API_KEY='your-mem0-api-key' +kimi +``` + +Inside Kimi Code, install the native plugin bundle and reload the session: + +```text +/plugins install https://github.com/mem0ai/mem0/tree/main/integrations/kimi-plugin +/reload +``` + +Run `/plugins info mem0` to confirm that the plugin, MCP server, skills, hooks, and sidekick agent loaded. + +## What you get + +- **Automatic capture:** Kimi records completed exchanges locally and flushes durable project knowledge to Mem0 in the background. +- **Automatic recall:** Relevant memories are added before Kimi answers the first prompt in a session. +- **Explicit search:** Kimi can call `search_memories` when it needs a more specific answer. +- **Six memory skills:** Search, status, remember, forget, pause, and resume use the same memory behavior as the other Mem0 coding-agent plugins. +- **Project scoping:** Memories stay attached to the repository, with separate personal and shared project lanes. +- **Kimi sidekick:** A focused subagent can investigate or implement a bounded task in a separate context. Filesystem isolation depends on the environment Kimi provides. + +Credentials are redacted before memory capture. If Mem0 is unavailable, hooks fail open so Kimi can continue its normal work. + +## How it works + +Kimi's native lifecycle events are translated into the shared Mem0 memory lifecycle: + +| Kimi event | What Mem0 does | +| --- | --- | +| `SessionStart` | Loads recent project context | +| `UserPromptSubmit` | Searches for relevant memories before the response | +| `PostToolUse` / `PostToolUseFailure` | Records useful tool results and failures | +| `Stop` | Captures the completed exchange | +| `PreCompact` / `SessionEnd` | Flushes pending capture in the background | +| `SubagentStart` / `SubagentStop` | Passes parent context to the Mem0 sidekick and records its lifecycle | + +## Verify the plugin + +In one session, say: + +```text +Remember exactly: the release codename for this repository is ORCHID-9274. +``` + +Start a new Kimi session in the same repository and ask: + +```text +What is the release codename for this repository? Do not infer it from repository files. +``` + +Kimi should return `ORCHID-9274` from memory. + +## Managing the plugin + +```text +/plugins list +/plugins info mem0 +/plugins disable mem0 +/plugins enable mem0 +/plugins remove mem0 +``` + +Run `/reload` or start a new session after enabling, disabling, or reinstalling the plugin. + +## Troubleshooting + +| Problem | Fix | +| --- | --- | +| Missing API key | Start Kimi from a shell where `MEM0_API_KEY` is exported. | +| Plugin changes do not appear | Run `/plugins reload`, then `/reload` or `/new`. | +| MCP server is disabled | Run `/plugins mcp enable mem0 mem0`, then `/reload`. | +| No memory in a later session | Wait a moment for the background flush, then ask Kimi to search memory explicitly. | +| Remove the plugin | Run `/plugins remove mem0`. | + + + + Add the same persistent project memory to Cursor + + + Use Mem0 with Claude Code and its isolated sidekick + + + + diff --git a/docs/integrations/openclaw.mdx b/docs/integrations/openclaw.mdx index f2fb49b1c..501900de6 100644 --- a/docs/integrations/openclaw.mdx +++ b/docs/integrations/openclaw.mdx @@ -5,6 +5,8 @@ description: "Add long-term memory to OpenClaw agents using the Mem0 plugin with Add long-term memory to [OpenClaw](https://github.com/openclaw/openclaw) agents with the `@mem0/openclaw-mem0` plugin. Your agent forgets everything between sessions. This plugin fixes that by automatically watching conversations, extracting what matters, and bringing it back when relevant. +Current package version: `1.1.0`. + ## Overview @@ -14,11 +16,12 @@ Add long-term memory to [OpenClaw](https://github.com/openclaw/openclaw) agents The plugin provides: 1. **Triage**: The agent extracts durable facts from conversations using a structured protocol with importance gates and domain overlays 2. **Recall**: Before each turn, relevant memories are retrieved with reranking and injected into context -3. **Dream**: Periodic memory consolidation merges duplicates, resolves conflicts, prunes stale entries -4. **Agent Tools**: Eight tools for explicit memory operations during conversations +3. **Agent Tools**: Eight tools for explicit memory operations during conversations Skills mode, `autoRecall`, and `autoCapture` are all enabled by default during `openclaw mem0 init`. +Automatic capture preserves the full text of selected user and assistant messages after noise filtering and secret redaction, without a per-message character cutoff. It selects recent messages and earlier work summaries; it does not capture every message in the conversation. + ## Requirements Check your OpenClaw version: @@ -101,7 +104,7 @@ You no longer need manual config editing to get started. Everything happens insi -That's it. No API key, no config file editing, no environment variables. The plugin is now active with skills-based memory (triage, recall, and dream) running automatically. +That's it. No API key, no config file editing, no environment variables. The plugin is now active with skills-based memory (triage and recall) running automatically. The chat flow uses the same underlying config as manual setup: it writes `apiKey`, `userId`, and `skills` config into `openclaw.json` for you. You can still open the file to inspect or override values afterward. @@ -142,7 +145,6 @@ That's it. No API key, no config file editing, no environment variables. The plu "keywordSearch": true, "identityAlwaysInclude": true }, - "dream": { "enabled": true }, "domain": "companion" } } @@ -441,7 +443,7 @@ If `openclaw plugins update` fails: ### Auto-Capture and Auto-Recall -Auto-capture and auto-recall are **enabled by default**. When skills mode is configured (the default after `openclaw mem0 init`), these are ignored in favor of the skills-based triage/recall/dream protocol. +Auto-capture and auto-recall are **enabled by default**. When skills mode is configured (the default after `openclaw mem0 init`), these are ignored in favor of the skills-based triage and recall protocol. To disable either: @@ -464,13 +466,12 @@ The agent can always use memory tools (`memory_add`, `memory_search`, etc.) expl ### Credential Protection -The plugin never stores API keys, tokens, or secrets as memories. Five independent layers enforce this: +The plugin never stores API keys, tokens, or secrets as memories. Four independent layers enforce this: 1. **Triage gate**: The extraction prompt rejects values matching known credential patterns (`sk-`, `m0-`, `ghp_`, `AKIA`, `Bearer`, `password=`, `token=`, `secret=`) -2. **Dream cleanup**: Periodic memory consolidation deletes any memories that slipped through containing credential patterns -3. **Extraction instructions**: Default extraction rules explicitly instruct the model to store only that a credential was configured, never the value -4. **Configurable patterns**: Add custom credential patterns via `skills.triage.credentialPatterns` -5. **CLI redaction**: `openclaw mem0 config show` redacts sensitive fields (`apiKey`, `oss.*.config.apiKey`) +2. **Extraction instructions**: Default extraction rules explicitly instruct the model to store only that a credential was configured, never the value +3. **Configurable patterns**: Add custom credential patterns via `skills.triage.credentialPatterns` +4. **CLI redaction**: `openclaw mem0 config show` redacts sensitive fields (`apiKey`, `oss.*.config.apiKey`) ### API Key Storage diff --git a/docs/integrations/opencode.mdx b/docs/integrations/opencode.mdx index 8593232fb..6c6ff5470 100644 --- a/docs/integrations/opencode.mdx +++ b/docs/integrations/opencode.mdx @@ -5,6 +5,8 @@ description: "Add persistent memory to OpenCode with the Mem0 plugin: native SDK Add persistent memory to [**OpenCode**](https://opencode.ai) with the Mem0 plugin. Your agent forgets everything between sessions. Mem0 fixes that by storing decisions, preferences, and learnings so they carry over automatically. +Current package version: `0.3.0`. + ## Prerequisites 1. A Mem0 API key (starts with `m0-`): @@ -33,7 +35,7 @@ opencode plugin @mem0/opencode-plugin **Or let your agent do it**: paste this into OpenCode: ``` -Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/integrations/mem0-plugin/.opencode-plugin/README.md +Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/integrations/opencode-plugin/README.md ``` This adds the plugin to your `~/.config/opencode/opencode.json`. Restart OpenCode. You get the native memory tools, lifecycle hooks, and all `/mem0-*` slash commands. The memory tools are registered by the plugin itself via the `mem0ai` SDK. No MCP server to configure. @@ -61,9 +63,9 @@ If you only need the memory tools without the plugin's hooks or skills, point Op | Component | Plugin (A) | Standalone MCP (B) | |-----------|:----------:|:------------------:| -| 9 memory tools | Native (SDK) | Remote MCP server | +| 10 memory tools | Native (SDK) | 9 remote MCP tools | | Lifecycle Hooks | Yes | No | -| 9 Skills | Yes | No | +| 7 Skills | Yes | No | ## Available Memory Tools @@ -78,6 +80,7 @@ If you only need the memory tools without the plugin's hooks or skills, point Op | `delete_all_memories` | Bulk delete all memories in scope | | `delete_entities` | Delete a user/agent/app/run entity and its memories | | `list_entities` | List users/agents/apps/runs stored in Mem0 | +| `get_event_status` | Check the processing status of an asynchronous memory event | ## Memory scope @@ -87,9 +90,9 @@ If you only need the memory tools without the plugin's hooks or skills, point Op |-------|-------|--------| | `project` *(default)* | this repo (`user_id` + `app_id`) | this repo | | `session` | this run only (`+ run_id`) | this run | -| `global` | **all your projects in the workspace** (`app_id: "*"`) | user-wide | +| `global` | **all your projects in the workspace** (filtered by your user ID) | user-wide | -Ask naturally, for example, *"search my memories across all my projects"*. The agent passes `scope: "global"`. For normal questions it stays scoped to the current project automatically. +Select `/mem0-scope global` before requesting a cross-project tool operation. A tool cannot enable global scope by supplying an argument alone. Switch back with `/mem0-scope project` when finished. To change the **default** scope (used when no scope is passed), run the `/mem0-scope` skill: @@ -117,31 +120,12 @@ The plugin uses the [mem0ai](https://www.npmjs.com/package/mem0ai) TypeScript SD | `experimental.session.compacting` | **Compaction** | Stores session state memory, then injects prior memories into compaction context so nothing is lost | | `shell.env` | **Shell env** | Exports `MEM0_USER_ID`, `MEM0_APP_ID`, `MEM0_SESSION_ID`, and `MEM0_BRANCH` to all shell executions | -## Auto-dream (memory consolidation) - -The plugin can automatically consolidate stored memories by merging duplicates, dropping stale/sensitive entries, and rewriting vague ones. This keeps your memory set clean over time. It runs at most once per session, and only when **all** gates pass: - -- **Time**: at least `minHours` (default 24) since the last consolidation -- **Sessions**: at least `minSessions` (default 5) sessions since then -- **Memories**: at least `minMemories` (default 20) stored for the project - -A filesystem lock (`~/.mem0/mem0-dream.lock`) keeps two sessions from consolidating at once. Tune the thresholds with a `dream` block in `~/.mem0/settings.json`, or disable entirely with `MEM0_DREAM=false`: - -```json -{ - "dream": { "enabled": true, "auto": true, "minHours": 24, "minSessions": 5, "minMemories": 20 } -} -``` - -If auto-dream hasn't run yet, it's almost always because a gate hasn't been met (most often too few memories). Run `/mem0-status` to see the exact gate progress (e.g. `sessions 2/5, memories 3/20`), `/mem0-dream` to consolidate **now** regardless of the gates, or lower the thresholds above. - ## Troubleshooting - **No tools appearing**: Restart OpenCode after installing - **"Connection failed"**: Verify your key is set: `echo $MEM0_API_KEY` - **Plugin not loading**: Run `opencode plugin @mem0/opencode-plugin` again, then restart - **Hooks not firing**: Hooks require the plugin install (Option A). MCP-only installs don't include hooks. -- **Auto-dream never runs**: It's gated (time + sessions + memories). Run `/mem0-status` to see which gate is blocking, or `/mem0-dream` to consolidate now. - **Wrong project name / memories not found**: The project id comes from your git remote; launch OpenCode from inside the repo (not your home directory). Check the resolved id with `/mem0-status`. diff --git a/docs/integrations/pi-agent.mdx b/docs/integrations/pi-agent.mdx index e377131da..e743bc7a9 100644 --- a/docs/integrations/pi-agent.mdx +++ b/docs/integrations/pi-agent.mdx @@ -1,19 +1,20 @@ --- title: Pi Agent -description: "Add persistent memory to Pi Agent with the Mem0 plugin semantic search, auto-capture, and dream consolidation." +description: "Add persistent memory to Pi Agent with the Mem0 plugin, semantic search, and automatic capture." --- Add persistent memory to [**Pi Agent**](https://pi.dev) with `@mem0/pi-agent-plugin`. Your agent forgets everything between sessions. This plugin fixes that by automatically capturing knowledge from conversations, storing it in Mem0's cloud memory layer, and retrieving relevant context before every response. +Current package version: `0.3.0`. + ## Overview The plugin provides: 1. **Auto-capture**: Extracts durable facts from both user and assistant messages automatically 2. **Semantic recall**: Retrieves relevant memories via the `mem0_memory` tool before each response -3. **Dream consolidation**: Periodic maintenance: merges duplicates, resolves contradictions, prunes stale entries -4. **Monorepo-aware scoping**: Uses git root for project detection, consistent across subdirectories -5. **Confirmation dialogs**: Destructive commands ask before acting via Pi's built-in UI -6. **8 skills + 8 commands**: Essential memory management from slash commands and agent-guided workflows +3. **Monorepo-aware scoping**: Uses git root for project detection, consistent across subdirectories +4. **Confirmation dialogs**: Destructive commands ask before acting via Pi's built-in UI +5. **6 skills + 6 commands**: Essential memory management from slash commands and agent-guided workflows ## Prerequisites @@ -59,14 +60,7 @@ For advanced settings, create `~/.pi/agent/mem0-config.json`: "userId": "your-username", "autoCapture": true, "defaultScope": "project", - "searchThreshold": 0.3, - "dream": { - "enabled": true, - "auto": true, - "minHours": 24, - "minSessions": 5, - "minMemories": 20 - } + "searchThreshold": 0.3 } ``` @@ -76,12 +70,7 @@ For advanced settings, create `~/.pi/agent/mem0-config.json`: | `userId` | `string` | `$MEM0_USER_ID` or `"default"` | User identity for memory scoping | | `autoCapture` | `boolean` | `true` | Store facts from conversations automatically | | `defaultScope` | `string` | `"project"` | Default memory scope: `project`, `session`, or `global` | -| `searchThreshold` | `number` | `0.3` | Minimum similarity score (0–1) a memory must reach to count as a match for `/mem0-search`, `/mem0-forget`, and `/mem0-pin`, enforced on each result's relevance score. Raise it to be stricter; lower it if relevant results are missed. | -| `dream.enabled` | `boolean` | `true` | Enable dream consolidation | -| `dream.auto` | `boolean` | `true` | Auto-trigger dreams when thresholds are met | -| `dream.minHours` | `number` | `24` | Minimum hours between auto-dreams | -| `dream.minSessions` | `number` | `5` | Minimum sessions before first auto-dream | -| `dream.minMemories` | `number` | `20` | Minimum memories before auto-dream triggers | +| `searchThreshold` | `number` | `0.3` | Minimum similarity score (0–1) a memory must reach to count as a match for `/mem0-search` and `/mem0-forget`, enforced on each result's relevance score. Raise it to be stricter; lower it if relevant results are missed. | ## What's Included @@ -89,11 +78,10 @@ For advanced settings, create `~/.pi/agent/mem0-config.json`: | Component | Description | |-----------|-------------| | `mem0_memory` tool | Agent-callable tool for search, add, get_all, delete, delete_all | -| 8 slash commands | Essential memory management from the command line | -| 8 skills | Guide the agent on how to use each capability | +| 6 slash commands | Essential memory management from the command line | +| 6 skills | Guide the agent on how to use each capability | | Auto-capture | Extracts and stores facts on every `agent_end` event | | System prompt | Appends memory policy to every agent turn | -| Dream consolidation | Automated memory maintenance with session/time/count gates | ## Agent Tool @@ -119,8 +107,6 @@ Tool output is truncated to 200 lines / 50KB to prevent context overflow. | `/mem0-forget ` | Search and delete memories (with confirmation dialog) | | `/mem0-search ` | Semantic search across memories | | `/mem0-tour [scope]` | Browse all memories grouped by category | -| `/mem0-dream` | Consolidate: merge duplicates, prune stale, resolve contradictions | -| `/mem0-pin ` | Pin a memory to protect from dream pruning (preserves memory ID) | | `/mem0-scope ` | Change default scope for this session (project, session, global) | | `/mem0-status` | Connection health, identity, and memory count | @@ -136,23 +122,12 @@ Memories are scoped using Mem0's `user_id`, `app_id`, and `run_id` parameters: The `app_id` is auto-detected from the git repository root (`git rev-parse --show-toplevel`), so all subdirectories within a monorepo share the same memory pool. Falls back to the working directory name for non-git directories. The `run_id` is derived from Pi's session file path. -## Dream Consolidation - -### Confirmation Dialogs +## Confirmation Dialogs Destructive and mutating commands use Pi's built-in `ctx.ui.confirm()` dialog before acting: - `/mem0-forget` asks "Delete this memory?" before deleting a single match -- `/mem0-pin` asks "Pin this memory?" before modifying it -- Cancelling either operation is always safe. No changes are made - -### Pin - -`/mem0-pin` uses Mem0's `update()` API to prepend `[PINNED]` to the memory text. This preserves the original memory ID. There is no add+delete cycle that would lose history or change the UUID. - -### Dream Consolidation - -The plugin includes automated memory maintenance ("dream") that merges duplicates, resolves contradictions, and prunes stale entries. When enabled, dreams auto-trigger after enough sessions, time, and memories accumulate (configurable via `dream.*` settings). Run `/mem0-dream` to trigger consolidation manually at any time. Pinned memories (via `/mem0-pin`) are protected from pruning. +- Cancelling the operation is always safe. No changes are made. ## Example Workflow @@ -172,7 +147,6 @@ You: What do you know about my preferences? - **Extension not loading**: Check Pi startup output for errors. Run `pi -e ./src/entry.ts` from the plugin directory for verbose output - **Memories not capturing**: Verify `autoCapture` is `true` (default). Check `/mem0-status` for connection health - **Wrong project detected**: The plugin uses the git repository root as `app_id`. If not in a git repo, it falls back to the working directory name. Run `/mem0-status` to see the detected project -- **Dream not triggering**: All three gates must pass (time, sessions, memories). Use `/mem0-dream` to force it manually @@ -184,3 +158,5 @@ You: What do you know about my preferences? + +Global tool operations require `/mem0-scope global` or `defaultScope: "global"` in plugin configuration. A model-supplied `scope` argument cannot enable cross-project access on its own. Empty or wildcard user, project, and session identities are rejected. diff --git a/docs/introduction.mdx b/docs/introduction.mdx index 53d424630..094855ad3 100644 --- a/docs/introduction.mdx +++ b/docs/introduction.mdx @@ -63,7 +63,7 @@ mode: "custom" Add memory to your coding agent

- Plugins that let Claude Code, Cursor, Codex, and other harnesses remember your project. Opens the Claude Code guide. + Start with Claude Code, then choose the individual guide for Cursor, Codex, Kimi Code, OpenCode, OpenClaw, Pi Agent, DeepSeek Harness, or Antigravity.

diff --git a/docs/llms.txt b/docs/llms.txt index ae45ad462..0523c6ab9 100644 --- a/docs/llms.txt +++ b/docs/llms.txt @@ -256,12 +256,12 @@ If the user is on a pre-current major (Python < 2, TS < 3, or a Platform call st - [Camel AI](https://docs.mem0.ai/integrations/camel-ai) [Both]: Use when the user is on Camel AI. - [ChatDev](https://docs.mem0.ai/integrations/chatdev) [Platform]: Use when the user is on ChatDev. - [Hermes](https://docs.mem0.ai/integrations/hermes) [Both]: Use when the user is on Hermes. -- [Pi Agent](https://docs.mem0.ai/integrations/pi-agent) [Platform]: Use when adding persistent memory to Pi Agent with the Mem0 plugin. -- [DeepSeek Harness](https://docs.mem0.ai/integrations/deepseek-plugin) [Platform]: Use when adding persistent memory to the DeepSeek Harness (Cordis) agent via the Mem0 plugin. +- [Pi Agent](https://docs.mem0.ai/integrations/pi-agent) [Platform]: Use when adding automatic capture, prompt recall, scoped memory, and six memory commands to Pi Agent. +- [DeepSeek Harness](https://docs.mem0.ai/integrations/deepseek-plugin) [Platform]: Use when adding automatic recall, completed-turn capture, and native search/add tools to DeepSeek Harness. - [OpenAI Agents SDK](https://docs.mem0.ai/integrations/openai-agents-sdk) [Platform]: Use when the user is on the OpenAI Agents SDK. - [Google AI ADK](https://docs.mem0.ai/integrations/google-ai-adk) [Platform]: Use when the user is on Google's Agent Development Kit. - [Mastra](https://docs.mem0.ai/integrations/mastra) [Platform]: Use when the user is on Mastra (TypeScript). -- [OpenClaw](https://docs.mem0.ai/integrations/openclaw) [Both]: Use when wiring Mem0 into Claude Code or editors via OpenClaw. +- [OpenClaw](https://docs.mem0.ai/integrations/openclaw) [Both]: Use when adding persistent memory to OpenClaw agents with Mem0 Platform or a self-hosted backend. - [Vercel AI SDK](https://docs.mem0.ai/integrations/vercel-ai-sdk) [Both]: Use when the user is on the Vercel AI SDK. - [Vercel](https://docs.mem0.ai/integrations/vercel) [Platform]: Use when deploying on Vercel and installing Mem0 from the Vercel Marketplace. - [Strands Agents](https://docs.mem0.ai/integrations/strands) [Both]: Use when the user is on AWS Strands and wants a native MemoryStore. @@ -269,10 +269,11 @@ If the user is on a pre-current major (Python < 2, TS < 3, or a Platform call st ### AI Coding Tools - [Claude Code](https://docs.mem0.ai/integrations/claude-code) [Both]: Use when wiring memory into Claude Code. - [Claude.ai](https://docs.mem0.ai/integrations/claude-ai) [Platform]: Use when connecting Mem0 to Claude.ai (the hosted web app) via a custom remote MCP connector, or when Claude's native memory seems to be crowding out mem0 tool calls. -- [Cursor](https://docs.mem0.ai/integrations/cursor) [Platform]: Use when wiring memory into Cursor. -- [Codex](https://docs.mem0.ai/integrations/codex) [Platform]: Use when wiring memory into Codex / other editor assistants. -- [OpenCode](https://docs.mem0.ai/integrations/opencode) [Platform]: Use when wiring memory into OpenCode. -- [Antigravity](https://docs.mem0.ai/integrations/antigravity) [Platform]: Use when wiring memory into Google Antigravity. +- [Cursor](https://docs.mem0.ai/integrations/cursor) [Platform]: Use when adding lifecycle capture, explicit memory recall, six skills, and a sidekick to Cursor. +- [Codex](https://docs.mem0.ai/integrations/codex) [Platform]: Use when adding automatic capture and recall, six memory skills, and a search tool to OpenAI Codex. +- [Kimi Code](https://docs.mem0.ai/integrations/kimi) [Platform]: Use when adding persistent project memory to Kimi Code through its native plugin lifecycle. +- [OpenCode](https://docs.mem0.ai/integrations/opencode) [Platform]: Use when adding ten native SDK memory tools, automatic context, seven skills, and project scoping to OpenCode. +- [Antigravity](https://docs.mem0.ai/integrations/antigravity) [Platform]: Use when adding lifecycle capture, explicit recall, six memory skills, and a native sidekick to Google Antigravity. ### Voice & Real-time - [LiveKit](https://docs.mem0.ai/integrations/livekit) [Platform]: Use when building real-time voice/video with memory. @@ -417,22 +418,25 @@ Each subdirectory is a Claude Code Skill (`SKILL.md` + supporting assets). Load Source: https://github.com/mem0ai/mem0/tree/main/integrations/claude-code-plugin -The `integrations/claude-code-plugin/` directory is the Claude Code plugin (v0.3.0, installs as `mem0@mem0-plugins`). It captures evidence locally through lifecycle hooks, extracts memories in a detached background worker, and exposes a single local MCP tool, `search_memories`, plus six `/mem0:*` skills and the `mem0:sidekick` agent. Pure-stdlib Python, nothing to install. +The self-contained Claude Code plugin lives in `integrations/claude-code-plugin/` (v0.3.0, installs as `mem0@mem0-plugins`). It captures evidence locally through lifecycle hooks, extracts memories in a detached background worker, and exposes a single local MCP tool, `search_memories`, plus six `/mem0:*` skills and the unchanged `mem0:sidekick` agent. Pure-stdlib Python, nothing to install. -### Editor Plugin (shared glue) +### Coding-Agent Plugin Sources -Source: https://github.com/mem0ai/mem0/tree/main/integrations/mem0-plugin +Source: https://github.com/mem0ai/mem0/tree/main/integrations/agent-plugin-core -The `integrations/mem0-plugin/` directory provides MCP server connection, lifecycle hooks, and skill bundling for Cursor, Codex, Kimi, Antigravity, and OpenCode. It exposes 9 MCP tools: `add_memory`, `search_memories`, `get_memories`, `get_memory`, `update_memory`, `delete_memory`, `delete_all_memories`, `delete_entities`, `list_entities`. Claude Code moved to `integrations/claude-code-plugin/` in v0.3.0; do not run both at the same time. +The `integrations/agent-plugin-core/` directory is the single source for shared Python and TypeScript memory behavior. Native plugins live in their own sibling directories, while `integrations/mem0-agent-plugin/` is the single portable Agent Plugins v1 package. TypeScript integrations reuse the core's lifecycle, formatting, identity, scoping, and telemetry utilities while keeping their public packages and native host APIs unchanged. Editor-specific setup docs (already listed above under `## Integrations > AI Coding Tools`): - `integrations/claude-code` [Both] - `integrations/cursor` [Platform] - `integrations/codex` [Platform] +- `integrations/kimi` [Platform] - `integrations/opencode` [Platform] - `integrations/antigravity` [Platform] - `integrations/openclaw` [Both] +- `integrations/pi-agent` [Platform] +- `integrations/deepseek-plugin` [Platform] ### MCP Endpoints diff --git a/integrations/AGENTS.md b/integrations/AGENTS.md index 6083dcf6a..c129dd48b 100644 --- a/integrations/AGENTS.md +++ b/integrations/AGENTS.md @@ -1,21 +1,22 @@ # Integrations (`integrations/`) -Agent and editor integrations. Each subdirectory is self-contained: its own `package.json`, lockfile, build, and tests. **There is no shared toolchain.** Check the table before running anything. +Agent and editor integrations. Most packages are self-contained; coding-agent plugins share the code in `agent-plugin-core/`. Check the table before running anything. | Directory | Package | Build | Lint | Test | |-----------|---------|-------|------|------| | `vercel-ai-sdk/` | `@mem0/vercel-ai-provider` | tsup (CJS+ESM) | ESLint + Prettier | jest + vitest (edge/node) | | `openclaw/` | `@mem0/openclaw-mem0` | tsup (ESM) | none | vitest | -| `claude-code-plugin/` | Claude Code plugin, installs as `mem0@mem0-plugins` (v0.3.0) | none | ruff | pytest | -| `mem0-plugin/` | Cursor / Codex / Kimi / Antigravity / OpenCode plugin (legacy — Claude Code moved to `claude-code-plugin/`) | none | none | pytest | -| `mem0-plugin/.opencode-plugin/` | `@mem0/opencode-plugin` | Bun | none | tsc type-check | +| `agent-plugin-core/` | Shared Python/TypeScript behavior, skill templates, builds, and conformance | Python build script | ruff + tsc | pytest + node:test | +| `mem0-agent-plugin/` | One portable Agent Plugins v1 package | Python | ruff | shared conformance | +| `claude-code-plugin/`, `cursor-plugin/`, `codex-plugin/`, `kimi-plugin/`, `antigravity-plugin/` | Self-contained native plugins generated from the shared Python core | Python | ruff | pytest | +| `opencode-plugin/` | `@mem0/opencode-plugin` (Bun/TypeScript) | tsup (via Bun) | tsc | bun test | | `pi-agent-plugin/` | `@mem0/pi-agent-plugin` | tsup | none | vitest | | `deepseek-plugin/` | `@mem0/deepseek-plugin` | tsup (ESM) | none | vitest | | `n8n-nodes-mem0/` | `@mem0/n8n-nodes-mem0` | tsc | ESLint (n8n-nodes-base) | none | | `zapier-mem0/` | `@mem0/zapier` | tsc | none | offline unit tests + `zapier validate` | | `mem0-strands/` | `mem0-strands` (PyPI) | hatch | Ruff + mypy | pytest | -pnpm everywhere except `.opencode-plugin/` (Bun) and `mem0-strands/` (Python: pip / hatch). Never npm, never yarn. +pnpm for TypeScript packages except `opencode-plugin/` (Bun). `mem0-strands/` uses Python/pip/hatch. Never npm or yarn. ## Commands @@ -41,8 +42,8 @@ Run the type check after every TypeScript change: `pnpm run typecheck` or `tsc - ## What each one is - **`vercel-ai-sdk/`** wraps the Vercel AI SDK through a `createMem0` provider. Integrations for AI-SDK repos go through this wrapper, not raw `MemoryClient`. -- **`claude-code-plugin/`** is the Claude Code plugin (v0.3.0, installs as `mem0@mem0-plugins`): local evidence capture via lifecycle hooks, background memory extraction to the Mem0 Platform, a local `search_memories` MCP tool, six `/mem0:*` skills, and the `mem0:sidekick` agent. Pure-stdlib Python — no dependencies to install. Its `core/` + `adapters/claude/` split marks engine vs. harness glue; future per-harness plugins start by copying `core/` and keeping the contract tests verbatim (see its `docs/CONTRACT.md`). -- **`mem0-plugin/`** connects Cursor, Codex, Kimi, Antigravity, and OpenCode to the MCP server at `mcp.mem0.ai` and installs lifecycle hooks for automatic memory capture. Exposes 9 MCP tools: `add_memory`, `search_memories`, `get_memories`, `get_memory`, `update_memory`, `delete_memory`, `delete_all_memories`, `delete_entities`, `list_entities`. The Claude Code plugin moved to [`claude-code-plugin/`](claude-code-plugin/) in v0.3.0 (installs as `mem0@mem0-plugins`); do not run both at the same time. +- **`agent-plugin-core/`** owns the shared Python memory runtime, TypeScript lifecycle utilities, skill templates, builds, and conformance runner. Claude Code is the behavioral source of truth. Native manifests and adapters live in sibling plugin directories; do not hand-edit their generated `core/` or `skills/` trees. Build and validation details are in [`agent-plugin-core/README.md`](agent-plugin-core/README.md). +- **`opencode-plugin/`** is a Bun/TypeScript plugin for OpenCode (`@mem0/opencode-plugin` on npm). It registers Mem0 memory tools as an OpenCode plugin with its own skills and telemetry. - **`openclaw/`**, **`pi-agent-plugin/`**, **`deepseek-plugin/`** are editor and agent plugins with the same shape. `deepseek-plugin/` registers Mem0 search/add tools as a native DeepSeek Harness (Cordis) plugin. - **`n8n-nodes-mem0/`** is an n8n community node: add, search, get, update, delete. - **`zapier-mem0/`** is a Zapier Platform CLI app: add, search, get, delete. It deploys to Zapier, not npm, so it is **not** in the release router. Deploy it with `gh workflow run zapier-mem0-cd.yml --ref main` (needs the `ZAPIER_DEPLOY_KEY` secret). @@ -50,11 +51,11 @@ Run the type check after every TypeScript change: `pnpm run typecheck` or `tsc - ## Adding an integration -1. Create `integrations//` and build it there, self-contained. +1. For a native coding-agent host, add `integrations/-plugin/` with `plugin-build.json`, its manifest, and a thin adapter, then generate its shared runtime. Portable clients use the single `mem0-agent-plugin/` package. Independent TypeScript integrations stay self-contained and import shared lifecycle behavior from `agent-plugin-core/typescript/`. 2. If it publishes to a registry, set `repository.directory: "integrations/"` in `package.json` so npm provenance links to the right subdirectory. 3. Add `.github/workflows/-checks.yml` and `-cd.yml`. Use `integrations/` in the `paths:` trigger, `working-directory`, and `cache-dependency-path`. Register the release tag prefix in the `case` block in `release.yml`, keeping the bare `v*` arm last. **Workflow filenames are load-bearing:** npm OIDC trusted publishing is pinned to repository plus workflow filename. Renaming one breaks publishing. 4. Register the CI workflow in `ci-gate.yml`: a path filter under the `changes` job, a call job, and an entry in the gate job's `needs` list. -5. If it is a Claude Code or editor marketplace plugin, register its path in all five `marketplace.json` files: root, `.claude-plugin/`, `.cursor-plugin/`, `.codex-plugin/`, and `.agents/plugins/`. +5. If it is a Claude Code or editor marketplace plugin, register the generated native bundle path in the applicable marketplace files. Preserve the existing public plugin name. 6. Document it under `docs/integrations/` and add the page to `docs/docs.json` and `docs/llms.txt`. 7. Add rows to the table above and to the CI/CD tables in [`../.github/AGENTS.md`](../.github/AGENTS.md). diff --git a/integrations/agent-plugin-core/README.md b/integrations/agent-plugin-core/README.md new file mode 100644 index 000000000..b2bc6e5cc --- /dev/null +++ b/integrations/agent-plugin-core/README.md @@ -0,0 +1,102 @@ +# Mem0 agent plugin core + +This directory is the single source of shared memory behavior for Mem0 coding-agent plugins. Installable plugins remain ordinary sibling directories under `integrations/`. + +## Architecture + +```text +integrations/ +├── agent-plugin-core/ # Shared source; never installed as a plugin +│ ├── python/ # Claude-derived capture, recall, MCP, scoping, and telemetry +│ ├── typescript/ # Shared lifecycle, formatting, identity, scoping, and telemetry +│ ├── skills/ # The only source for the six generated memory skills +│ ├── build/ # Bundle builder, schemas, and validation +│ ├── conformance/ # One offline/live verification entry point +│ └── tests/ +├── mem0-agent-plugin/ # One portable Agent Plugins v1 package +├── claude-code-plugin/ # Native Claude package and adapter +├── cursor-plugin/ # Native Cursor package and adapter +├── codex-plugin/ # Native Codex package and adapter +├── kimi-plugin/ # Native Kimi package and adapter +└── antigravity-plugin/ # Native Antigravity package and adapter +``` + +Each native directory owns only its manifest, native hooks or adapter, tests, and `plugin-build.json`. Its `core/` and `skills/` directories are generated from this module. They are committed because clients install a self-contained plugin directory and the Agent Plugins specification forbids package files from resolving outside the plugin root. + +Claude Code remains the behavioral source of truth. Its sidekick stays at `claude-code-plugin/agents/sidekick.md` and is not generated or copied to hosts without a compatible native subagent interface. + +TypeScript integrations (`openclaw`, `opencode-plugin`, `pi-agent-plugin`, and `deepseek-plugin`) import `typescript/src/` at build time. Their package builders include the shared implementation in their normal output; they do not carry checked-in copies. + +## Build and verify + +From the repository root: + +```bash +python3.11 -m venv /tmp/mem0-agent-plugins +/tmp/mem0-agent-plugins/bin/pip install \ + -r integrations/agent-plugin-core/requirements-dev.txt + +for host in claude-code cursor codex kimi antigravity; do + /tmp/mem0-agent-plugins/bin/python \ + integrations/agent-plugin-core/build/build.py "$host" \ + --kind native --check +done + +/tmp/mem0-agent-plugins/bin/python \ + integrations/agent-plugin-core/build/build.py mem0-agent-plugin \ + --kind portable --check +``` + +Use `--sync` instead of `--check` after changing `python/` or `skills/`. This only replaces generated `core/` and `skills/` content; it does not change manifests, adapters, tests, or the Claude sidekick. + +Run every offline Python and TypeScript check and write one machine-readable report: + +```bash +/tmp/mem0-agent-plugins/bin/python \ + integrations/agent-plugin-core/conformance/run.py \ + --install \ + --report /tmp/mem0-plugin-conformance.json +``` + +For every TypeScript integration, this also builds the publishable package, verifies its required entry files, and rejects compiled artifacts that still import monorepo source. This keeps published plugins self-contained without committing their `dist/` directories. + +The offline suite does not contact Mem0 Platform. An explicit disposable key enables the inherited live scoping suite: + +```bash +export MEM0_API_KEY="m0-disposable-test-key" +/tmp/mem0-agent-plugins/bin/python \ + integrations/agent-plugin-core/conformance/run.py \ + --group live-platform --live \ + --report /tmp/mem0-plugin-live-conformance.json +``` + +Do not put a real key in source files, command history shared with others, or pull-request configuration. + +## Add a plugin + +For another native Python host: + +1. Add `integrations/-plugin/` with its native manifest and the smallest adapter that translates host events. +2. Add `plugin-build.json` declaring the plugin-root variable and runtime files. +3. Add one adapter contract test. +4. Register the host in `build/build.py` and `conformance/run.py`. +5. Run `--sync`, `--check`, and the conformance command above. + +Keep capture, recall, memory scoping, redaction, skill text, and telemetry in this shared module. Host directories should contain only behavior required by their native SDK. + +For a TypeScript host, import the shared lifecycle modules directly and keep only native SDK registration in the integration. Do not advertise capture, compaction, or sidekick behavior unless the host exposes the necessary lifecycle seam. + +## Host capture capabilities + +| Host | Conversation capture | Tool outcomes | Subagent context and correlation | +| --- | --- | --- | --- | +| Claude Code | Incremental active transcript branch | Native success/failure hooks | Parent context; native agent ID | +| Cursor | Prompt and response hooks; duplicate responses suppressed | Native success/failure hooks | Sidekick searches itself; no parent-injection claim | +| Codex | Native prompt and final-response fields | Structured failure indicators when present; otherwise unknown | Parent context; native agent ID | +| Kimi | Prompt hooks and completed v2 wire output | Native success/failure hooks | Parent context; without an ID, a stop matches only one unambiguous active run | +| Antigravity | Incremental completed transcript messages, including later prompts | Native tool errors | Sidekick searches itself; no worktree-isolation claim | +| Portable v1 | Explicit memory skills | No native lifecycle hooks | No native subagent declaration | + +An uncorrelated subagent completion is kept as its own record; the plugin never guesses which overlapping run completed. Codex's documented hook fields already match the shared input contract, so no speculative field aliases or unsupported failure event are registered. + +Offline conformance exercises the adapters and MCP servers with native-shaped payloads and builds each distributable package. It does not establish that every installed editor or Harness version loads the plugin correctly; those checks require smoke tests in the actual hosts. diff --git a/integrations/agent-plugin-core/build/__init__.py b/integrations/agent-plugin-core/build/__init__.py new file mode 100644 index 000000000..81c68c34c --- /dev/null +++ b/integrations/agent-plugin-core/build/__init__.py @@ -0,0 +1 @@ +"""Build and validate self-contained Mem0 agent plugins.""" diff --git a/integrations/agent-plugin-core/build/build.py b/integrations/agent-plugin-core/build/build.py new file mode 100644 index 000000000..7a72e5c51 --- /dev/null +++ b/integrations/agent-plugin-core/build/build.py @@ -0,0 +1,252 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import re +import shutil +import tempfile +from collections.abc import Mapping +from pathlib import Path + +try: + from .validate import validate_bundle +except ImportError: + from validate import validate_bundle + +CORE_ROOT = Path(__file__).resolve().parents[1] +REPOSITORY_ROOT = CORE_ROOT.parents[1] +INTEGRATIONS_ROOT = REPOSITORY_ROOT / "integrations" +SHARED_SKILLS = CORE_ROOT / "skills" +PORTABLE_PLUGIN = "mem0-agent-plugin" +NATIVE_PLUGINS = { + "claude-code": INTEGRATIONS_ROOT / "claude-code-plugin", + "cursor": INTEGRATIONS_ROOT / "cursor-plugin", + "codex": INTEGRATIONS_ROOT / "codex-plugin", + "kimi": INTEGRATIONS_ROOT / "kimi-plugin", + "antigravity": INTEGRATIONS_ROOT / "antigravity-plugin", +} +PROTECTED_OUTPUTS = { + REPOSITORY_ROOT, + INTEGRATIONS_ROOT, + CORE_ROOT, + INTEGRATIONS_ROOT / PORTABLE_PLUGIN, + *NATIVE_PLUGINS.values(), +} +TEMPLATE_TOKEN = re.compile(r"{{([A-Z_]+)}}") +TEMPLATE_TOKENS = { + "PLUGIN_ROOT", + "PLUGIN_DATA", + "PLUGIN_DATA_ARG", + "COMMAND_PREFIX", + "HARNESS_ID", + "HARNESS_NAME", +} + + +def render_template(source: str, values: Mapping[str, str]) -> str: + def replace(match: re.Match[str]) -> str: + token = match.group(1) + if token not in TEMPLATE_TOKENS or token not in values: + raise ValueError(f"unknown or unresolved template token: {token}") + return values[token] + + rendered = TEMPLATE_TOKEN.sub(replace, source) + if "{{" in rendered or "}}" in rendered: + raise ValueError("unresolved template token") + return rendered + + +def replace_output(staged: Path, output: Path) -> Path: + """Replace one explicit build output without touching its siblings.""" + if not staged.is_dir(): + raise ValueError(f"staged directory does not exist: {staged}") + + resolved_output = output.resolve() + if resolved_output in {path.resolve() for path in PROTECTED_OUTPUTS}: + raise ValueError(f"refusing protected output path: {output}") + if output.exists() and not output.is_dir(): + raise ValueError(f"output path is not a directory: {output}") + + output.parent.mkdir(parents=True, exist_ok=True) + temporary = Path(tempfile.mkdtemp(prefix=f".{output.name}-", dir=output.parent)) + try: + shutil.copytree(staged, temporary, dirs_exist_ok=True) + if output.exists(): + shutil.rmtree(output) + temporary.replace(output) + finally: + if temporary.exists(): + shutil.rmtree(temporary) + return output + + +def _bundle_python( + staged: Path, + host: str, + plugin_root: str, + *, + plugin_data: str = "", + portable: bool = False, +) -> None: + core = staged / "core" + core.mkdir() + for source in sorted((CORE_ROOT / "python").glob("*.py")): + if portable and source.name in {"flush_worker.py", "hook_runner.py"}: + continue + shutil.copy2(source, core / source.name) + + values = { + "PLUGIN_ROOT": plugin_root, + "PLUGIN_DATA": "${PLUGIN_DATA}", + "PLUGIN_DATA_ARG": f'--plugin-data-dir "{plugin_data}"' if plugin_data else "", + "COMMAND_PREFIX": "mem0", + "HARNESS_ID": host, + "HARNESS_NAME": host.replace("-", " ").title(), + } + for source in sorted(SHARED_SKILLS.glob("*/SKILL.md.tmpl")): + target = staged / "skills" / source.parent.name / "SKILL.md" + target.parent.mkdir(parents=True) + rendered = render_template(source.read_text(encoding="utf-8"), values) + if portable: + rendered = "\n".join( + line + for line in rendered.splitlines() + if not line.startswith(("argument-hint:", "disable-model-invocation:")) + ) + "\n" + target.write_text(rendered, encoding="utf-8") + + +def _build_portable(staged: Path) -> None: + source = INTEGRATIONS_ROOT / PORTABLE_PLUGIN + _copy_declared_files(staged, source, {"plugin.json": "plugin.json", "mcp.json": "mcp.json"}) + _bundle_python(staged, "coding-agent", "${PLUGIN_ROOT}", portable=True) + + +def _copy_declared_files(staged: Path, source_root: Path, files: object) -> None: + if not isinstance(files, dict): + raise ValueError("native files must be an object") + source_root = source_root.resolve() + staged_root = staged.resolve() + for source_name, target_name in files.items(): + if not isinstance(source_name, str) or not isinstance(target_name, str): + raise ValueError("native file paths must be strings") + source = (source_root / source_name).resolve() + target = (staged / target_name).resolve() + if not source.is_relative_to(source_root) or not target.is_relative_to(staged_root): + raise ValueError("native file paths must stay inside their roots") + if not source.is_file(): + raise ValueError(f"native source file does not exist: {source_name}") + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(source, target) + + +def _build_native(host: str, source_root: Path, staged: Path, descriptor: dict) -> None: + native = descriptor.get("native") + if not isinstance(native, dict) or not isinstance(native.get("pluginRoot"), str): + raise ValueError(f"native build is not declared for {host}") + _bundle_python( + staged, + host, + native["pluginRoot"], + plugin_data=str(native.get("pluginData") or ""), + ) + _copy_declared_files(staged, source_root, native.get("files", {})) + + +def build(host: str, kind: str, output: Path) -> Path: + if kind not in {"portable", "native"}: + raise ValueError(f"unknown bundle kind: {kind}") + if kind == "portable": + if host != PORTABLE_PLUGIN: + raise ValueError(f"the portable bundle is {PORTABLE_PLUGIN}") + source_root = INTEGRATIONS_ROOT / PORTABLE_PLUGIN + descriptor = None + else: + source_root = NATIVE_PLUGINS.get(host) + if source_root is None: + raise ValueError(f"unknown host: {host}") + descriptor_path = source_root / "plugin-build.json" + descriptor = json.loads(descriptor_path.read_text(encoding="utf-8")) + with tempfile.TemporaryDirectory(prefix=f"mem0-{host}-{kind}-") as temporary: + staged = Path(temporary) / "bundle" + staged.mkdir() + if kind == "portable": + _build_portable(staged) + else: + assert descriptor is not None + _build_native(host, source_root, staged, descriptor) + errors = validate_bundle(staged, kind) + if errors: + raise ValueError("invalid bundle:\n" + "\n".join(errors)) + return replace_output(staged, output) + + +def installable_root(host: str, kind: str) -> Path: + if kind == "portable" and host == PORTABLE_PLUGIN: + return INTEGRATIONS_ROOT / PORTABLE_PLUGIN + if kind == "native" and host in NATIVE_PLUGINS: + return NATIVE_PLUGINS[host] + raise ValueError(f"unknown {kind} plugin: {host}") + + +def bundle_drift(host: str, kind: str) -> list[str]: + target = installable_root(host, kind) + with tempfile.TemporaryDirectory(prefix=f"mem0-check-{host}-") as temporary: + generated = build(host, kind, Path(temporary) / "bundle") + errors: list[str] = [] + for directory in ("core", "skills"): + expected = { + path.relative_to(generated) + for path in (generated / directory).rglob("*") + if path.is_file() + } + actual = { + path.relative_to(target) + for path in (target / directory).rglob("*") + if path.is_file() and "__pycache__" not in path.parts and path.suffix != ".pyc" + } + errors.extend(f"missing generated file: {path}" for path in sorted(expected - actual)) + errors.extend(f"stale generated file: {path}" for path in sorted(actual - expected)) + for source in sorted(path for path in generated.rglob("*") if path.is_file()): + relative = source.relative_to(generated) + installed = target / relative + if not installed.is_file() or source.read_bytes() != installed.read_bytes(): + errors.append(f"generated file differs: {relative}") + return errors + + +def sync_generated(host: str, kind: str) -> Path: + target = installable_root(host, kind) + with tempfile.TemporaryDirectory(prefix=f"mem0-sync-{host}-") as temporary: + generated = build(host, kind, Path(temporary) / "bundle") + for directory in ("core", "skills"): + replace_output(generated / directory, target / directory) + return target + + +def main() -> int: + parser = argparse.ArgumentParser(description="Build a self-contained Mem0 agent plugin") + parser.add_argument("host") + parser.add_argument("--kind", choices=("portable", "native"), required=True) + action = parser.add_mutually_exclusive_group(required=True) + action.add_argument("--output", type=Path) + action.add_argument("--check", action="store_true") + action.add_argument("--sync", action="store_true") + args = parser.parse_args() + if args.check: + errors = bundle_drift(args.host, args.kind) + if errors: + print("\n".join(errors)) + return 1 + print(f"Current {args.kind} bundle: {installable_root(args.host, args.kind)}") + elif args.sync: + print(sync_generated(args.host, args.kind)) + else: + assert args.output is not None + print(build(args.host, args.kind, args.output)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/agent-plugin-core/build/schemas/mcp.schema.json b/integrations/agent-plugin-core/build/schemas/mcp.schema.json new file mode 100644 index 000000000..1c0626056 --- /dev/null +++ b/integrations/agent-plugin-core/build/schemas/mcp.schema.json @@ -0,0 +1,89 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://agent-plugins.org/schemas/1.0.0/mcp.schema.json", + "title": "Agent Plugins MCP Configuration", + "description": "Machine-readable schema for mcp.json in Agent Plugins 1.0.0. The Agent Plugins specification defines additional semantic and operational requirements.", + "type": "object", + "properties": { + "$schema": { + "const": "https://agent-plugins.org/schemas/1.0.0/mcp.schema.json", + "description": "Canonical identifier of the MCP configuration schema for the Agent Plugins version targeted by this document." + }, + "mcpServers": { + "type": "object", + "additionalProperties": { "$ref": "#/$defs/server" } + } + }, + "required": ["$schema", "mcpServers"], + "additionalProperties": false, + "$defs": { + "server": { + "title": "MCP server", + "oneOf": [ + { "$ref": "#/$defs/stdioServer" }, + { "$ref": "#/$defs/streamableHttpServer" }, + { "$ref": "#/$defs/sseServer" } + ] + }, + "stdioServer": { + "title": "stdio MCP server", + "type": "object", + "properties": { + "type": { "const": "stdio" }, + "command": { + "type": "string", + "minLength": 1, + "description": "Executable token. Resolution rules are defined by the Agent Plugins specification." + }, + "args": { "type": "array", "items": { "type": "string" } }, + "env": { + "type": "object", + "propertyNames": { "not": { "enum": ["PLUGIN_ROOT", "PLUGIN_DATA"] } }, + "additionalProperties": { "type": "string" } + }, + "cwd": { + "type": "string", + "pattern": "^(?:\\./|\\$\\{PLUGIN_ROOT\\}(?:/|$)|\\$\\{PLUGIN_DATA\\}(?:/|$))", + "description": "Plugin-relative, PLUGIN_ROOT-rooted, or PLUGIN_DATA-rooted working directory. Filesystem containment is validated separately." + } + }, + "required": ["type", "command"], + "additionalProperties": false + }, + "streamableHttpServer": { + "title": "Streamable HTTP MCP server", + "type": "object", + "properties": { + "type": { "const": "streamable-http" }, + "url": { + "type": "string", + "minLength": 1, + "description": "MCP endpoint URL. URL semantics are defined by the Agent Plugins specification." + }, + "headers": { "$ref": "#/$defs/headers" } + }, + "required": ["type", "url"], + "additionalProperties": false + }, + "sseServer": { + "title": "Legacy HTTP+SSE MCP server", + "type": "object", + "properties": { + "type": { "const": "sse" }, + "url": { + "type": "string", + "minLength": 1, + "description": "MCP endpoint URL. URL semantics are defined by the Agent Plugins specification." + }, + "headers": { "$ref": "#/$defs/headers" } + }, + "required": ["type", "url"], + "additionalProperties": false + }, + "headers": { + "title": "HTTP headers", + "type": "object", + "additionalProperties": { "type": "string" } + } + } +} diff --git a/integrations/agent-plugin-core/build/schemas/plugin.schema.json b/integrations/agent-plugin-core/build/schemas/plugin.schema.json new file mode 100644 index 000000000..fc3155c35 --- /dev/null +++ b/integrations/agent-plugin-core/build/schemas/plugin.schema.json @@ -0,0 +1,45 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", + "title": "Agent Plugins Manifest", + "description": "Machine-readable schema for plugin.json in Agent Plugins 1.0.0. The Agent Plugins specification defines additional semantic and operational requirements.", + "type": "object", + "properties": { + "$schema": { + "const": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", + "description": "Canonical identifier of the plugin manifest schema for the Agent Plugins version targeted by this document." + }, + "name": { + "type": "string", + "minLength": 1, + "maxLength": 64, + "pattern": "^(?!.*(?:--|\\.\\.))[a-z0-9](?:[a-z0-9.-]*[a-z0-9])?$", + "description": "Human-readable plugin name." + }, + "version": { "type": "string" }, + "description": { "type": "string" }, + "author": { + "type": "object", + "properties": { + "name": { "type": "string" }, + "email": { "type": "string" }, + "url": { "type": "string" } + }, + "additionalProperties": false + }, + "homepage": { "type": "string" }, + "repository": { "type": "string" }, + "license": { "type": "string" }, + "keywords": { + "type": "array", + "items": { "type": "string" } + }, + "extensions": { + "type": "object", + "description": "Client-specific manifest data keyed by reverse-domain extension namespace. Agent Plugins assigns no semantics to namespace object contents.", + "additionalProperties": { "type": "object" } + } + }, + "required": ["$schema", "name"], + "additionalProperties": false +} diff --git a/integrations/agent-plugin-core/build/validate.py b/integrations/agent-plugin-core/build/validate.py new file mode 100644 index 000000000..fea164e16 --- /dev/null +++ b/integrations/agent-plugin-core/build/validate.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +from jsonschema import Draft202012Validator +from skills_ref import validate as validate_skill + +SCHEMAS = Path(__file__).resolve().parent / "schemas" + + +def _schema_errors(path: Path, schema_name: str) -> list[str]: + if not path.is_file(): + return [f"{path.name}: file is required"] + + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as error: + return [f"{path.name}: {error}"] + + schema = json.loads((SCHEMAS / schema_name).read_text(encoding="utf-8")) + errors = Draft202012Validator(schema).iter_errors(value) + return [ + f"{path.name}{''.join(f'.{part}' for part in error.absolute_path)}: {error.message}" + for error in sorted(errors, key=lambda item: tuple(str(part) for part in item.absolute_path)) + ] + + +def validate_bundle(root: Path, kind: str) -> list[str]: + errors: list[str] = [] + if not root.is_dir(): + return [f"{root}: directory is required"] + + for path in sorted(root.rglob("*")): + if path.is_symlink(): + errors.append(f"{path.relative_to(root)}: symlinks are not allowed in release bundles") + + if kind == "portable": + errors.extend(_schema_errors(root / "plugin.json", "plugin.schema.json")) + if (root / "mcp.json").exists(): + errors.extend(_schema_errors(root / "mcp.json", "mcp.schema.json")) + skills = root / "skills" + if skills.is_dir(): + for skill in sorted(path for path in skills.iterdir() if path.is_dir()): + errors.extend(f"skills/{skill.name}: {error}" for error in validate_skill(skill)) + elif kind == "native": + for path in sorted(root.rglob("*.json")): + relative = path.relative_to(root) + try: + json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as error: + errors.append(f"{relative}: {error}") + return sorted(errors) + + +def main() -> int: + parser = argparse.ArgumentParser(description="Validate a generated Mem0 plugin bundle") + parser.add_argument("root", type=Path) + parser.add_argument("--kind", choices=("portable", "native"), required=True) + args = parser.parse_args() + + errors = validate_bundle(args.root, args.kind) + if errors: + for error in errors: + print(error, file=sys.stderr) + return 1 + + print(f"Validated {args.kind} bundle: {args.root}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/agent-plugin-core/conformance/artifacts.py b/integrations/agent-plugin-core/conformance/artifacts.py new file mode 100644 index 000000000..a7aa6efa7 --- /dev/null +++ b/integrations/agent-plugin-core/conformance/artifacts.py @@ -0,0 +1,58 @@ +#!/usr/bin/env python3 +"""Verify that a built TypeScript plugin is self-contained and publishable.""" + +from __future__ import annotations + +import argparse +import re +import time +from pathlib import Path +from typing import Any + + +CORE_ROOT = Path(__file__).resolve().parents[1] +INTEGRATIONS_ROOT = CORE_ROOT.parent +TYPESCRIPT_ARTIFACTS = { + "openclaw": (INTEGRATIONS_ROOT / "openclaw", ("dist/index.js", "dist/index.d.ts")), + "opencode": (INTEGRATIONS_ROOT / "opencode-plugin", ("dist/index.js", "index.d.ts")), + "pi-agent": ( + INTEGRATIONS_ROOT / "pi-agent-plugin", + ("dist/index.js", "dist/index.d.ts", "dist/entry.js", "dist/entry.d.ts"), + ), + "deepseek": (INTEGRATIONS_ROOT / "deepseek-plugin", ("dist/index.js", "dist/index.d.ts")), +} +MONOREPO_IMPORT = re.compile( + r"(?:from\s+|import\s*\(|require\s*\()\s*['\"][^'\"]*agent-plugin-core" +) + + +def verify_artifact(group: str, package: Path, required: tuple[str, ...]) -> dict[str, Any]: + started = time.monotonic() + errors = [f"missing package artifact: {name}" for name in required if not (package / name).is_file()] + dist = package / "dist" + for pattern in ("*.js", "*.mjs", "*.cjs", "*.d.ts"): + for artifact in dist.rglob(pattern) if dist.is_dir() else (): + if MONOREPO_IMPORT.search(artifact.read_text(encoding="utf-8")): + errors.append(f"monorepo source import in package artifact: {artifact.relative_to(package)}") + return { + "name": f"{group}-artifact", + "group": group, + "status": "failed" if errors else "passed", + "duration_seconds": round(time.monotonic() - started, 3), + **({"output": "\n".join(errors)} if errors else {}), + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("group", choices=tuple(TYPESCRIPT_ARTIFACTS)) + args = parser.parse_args() + package, required = TYPESCRIPT_ARTIFACTS[args.group] + result = verify_artifact(args.group, package, required) + if result.get("output"): + print(result["output"]) + return 0 if result["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/agent-plugin-core/conformance/run.py b/integrations/agent-plugin-core/conformance/run.py new file mode 100644 index 000000000..86f7de10f --- /dev/null +++ b/integrations/agent-plugin-core/conformance/run.py @@ -0,0 +1,320 @@ +#!/usr/bin/env python3 +"""Run every coding-agent plugin check and emit one conformance report.""" + +from __future__ import annotations + +import argparse +import json +import os +import subprocess +import sys +import tempfile +import time +from pathlib import Path +from typing import Any + + +CORE_ROOT = Path(__file__).resolve().parents[1] +REPOSITORY_ROOT = CORE_ROOT.parents[1] +PYTHON_HOSTS = ("claude-code", "cursor", "codex", "kimi", "antigravity") +GROUPS = ( + "python-bundles", + "python-tests", + "typescript-core", + "openclaw", + "opencode", + "pi-agent", + "deepseek", +) +LIVE_GROUP = "live-platform" + +sys.path.insert(0, str(CORE_ROOT)) +sys.path.insert(0, str(CORE_ROOT / "python")) +from memory_core import redact # noqa: E402 +from build.build import build # noqa: E402 +from conformance.artifacts import TYPESCRIPT_ARTIFACTS, verify_artifact as _typescript_artifact_check # noqa: E402 + + +def _package_directories() -> dict[str, Path]: + return { + "typescript-core": CORE_ROOT / "typescript", + "openclaw": REPOSITORY_ROOT / "integrations" / "openclaw", + "opencode": REPOSITORY_ROOT / "integrations" / "opencode-plugin", + "pi-agent": REPOSITORY_ROOT / "integrations" / "pi-agent-plugin", + "deepseek": REPOSITORY_ROOT / "integrations" / "deepseek-plugin", + } + + +def _runtime_commands() -> dict[str, list[list[str]]]: + return { + "python-tests": [ + [ + sys.executable, + "-m", + "pytest", + "integrations/agent-plugin-core/tests", + "integrations/claude-code-plugin/tests", + "integrations/cursor-plugin/tests", + "integrations/codex-plugin/tests", + "integrations/kimi-plugin/tests", + "integrations/antigravity-plugin/tests", + "-q", + "--ignore=integrations/agent-plugin-core/tests/test_conformance.py", + "--ignore=integrations/claude-code-plugin/tests/integration", + ] + ], + "typescript-core": [["pnpm", "test"], ["pnpm", "typecheck"]], + "openclaw": [ + ["pnpm", "test"], + ["pnpm", "exec", "tsc", "--noEmit"], + ["pnpm", "build"], + ], + "opencode": [ + ["bun", "test"], + ["bun", "run", "type-check"], + ["bun", "run", "build"], + ], + "pi-agent": [ + ["pnpm", "test"], + ["pnpm", "typecheck"], + ["pnpm", "build"], + ], + "deepseek": [ + ["pnpm", "test"], + ["pnpm", "typecheck"], + ["pnpm", "build"], + ], + LIVE_GROUP: [ + [ + sys.executable, + "-m", + "pytest", + "integrations/claude-code-plugin/tests/integration", + "-q", + ] + ], + } + + +def _planned_checks(groups: set[str]) -> list[dict[str, Any]]: + checks: list[dict[str, Any]] = [] + if "python-bundles" in groups: + checks.append( + { + "name": "python-bundles", + "group": "python-bundles", + "status": "planned", + "hosts": list(PYTHON_HOSTS), + "kinds": {"native": list(PYTHON_HOSTS), "portable": ["mem0-agent-plugin"]}, + } + ) + for group, commands in _runtime_commands().items(): + if group not in groups: + continue + for index, command in enumerate(commands, start=1): + checks.append( + { + "name": f"{group}-{index}", + "group": group, + "status": "planned", + "command": command, + } + ) + if group in TYPESCRIPT_ARTIFACTS: + _, required = TYPESCRIPT_ARTIFACTS[group] + checks.append( + { + "name": f"{group}-artifact", + "group": group, + "status": "planned", + "required": list(required), + } + ) + return checks + + +def _command_check( + name: str, + group: str, + command: list[str], + *, + cwd: Path, + environment_overrides: dict[str, str] | None = None, +) -> dict[str, Any]: + started = time.monotonic() + environment = dict(os.environ) + environment.update(environment_overrides or {}) + safe_command = [redact(part) for part in command] + try: + result = subprocess.run( + command, + cwd=cwd, + env=environment, + text=True, + capture_output=True, + check=False, + ) + output = redact((result.stdout + result.stderr).strip()) + return { + "name": name, + "group": group, + "status": "passed" if result.returncode == 0 else "failed", + "duration_seconds": round(time.monotonic() - started, 3), + "command": safe_command, + "exit_code": result.returncode, + **({"output": output[-8_000:]} if output else {}), + } + except OSError as exc: + return { + "name": name, + "group": group, + "status": "failed", + "duration_seconds": round(time.monotonic() - started, 3), + "command": safe_command, + "exit_code": None, + "output": redact(str(exc)), + } + + +def _bundle_checks(artifacts_dir: Path) -> list[dict[str, Any]]: + checks: list[dict[str, Any]] = [] + plugins = [(host, "native") for host in PYTHON_HOSTS] + [("mem0-agent-plugin", "portable")] + for host, kind in plugins: + started = time.monotonic() + try: + output = artifacts_dir / host + build(host, kind, output) + status, error = "passed", "" + except Exception as exc: + status, error = "failed", f"{type(exc).__name__}: {exc}" + checks.append( + { + "name": f"{host}-{kind}", + "group": "python-bundles", + "host": host, + "kind": kind, + "status": status, + "duration_seconds": round(time.monotonic() - started, 3), + **({"output": error} if error else {}), + } + ) + return checks + + +def _install_checks(groups: set[str]) -> list[dict[str, Any]]: + checks: list[dict[str, Any]] = [] + for group, package in _package_directories().items(): + if group not in groups: + continue + command = ( + ["bun", "install", "--frozen-lockfile"] + if group == "opencode" + else ["pnpm", "install", "--frozen-lockfile"] + ) + checks.append( + _command_check( + f"{group}-install", + group, + command, + cwd=package, + environment_overrides={"CI": "true"}, + ) + ) + return checks + + +def _runtime_checks(groups: set[str]) -> list[dict[str, Any]]: + checks: list[dict[str, Any]] = [] + directories = { + **_package_directories(), + "python-tests": REPOSITORY_ROOT, + LIVE_GROUP: REPOSITORY_ROOT, + } + for group, commands in _runtime_commands().items(): + if group not in groups: + continue + for index, command in enumerate(commands, start=1): + checks.append( + _command_check( + f"{group}-{index}", + group, + command, + cwd=directories[group], + ) + ) + if group in TYPESCRIPT_ARTIFACTS: + _, required = TYPESCRIPT_ARTIFACTS[group] + checks.append( + _typescript_artifact_check( + group, + directories[group], + required, + ) + ) + return checks + + +def _print_report(report: dict[str, Any]) -> None: + for check in report["checks"]: + marker = {"passed": "PASS", "failed": "FAIL", "planned": "PLAN"}[check["status"]] + duration = f" ({check['duration_seconds']:.3f}s)" if "duration_seconds" in check else "" + print(f"{marker:4} {check['name']}{duration}") + if check["status"] == "failed" and check.get("output"): + print(check["output"]) + print(f"\nConformance: {report['status']} ({len(report['checks'])} checks)") + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--group", action="append", choices=("all", *GROUPS, LIVE_GROUP), default=[]) + parser.add_argument("--list", action="store_true", help="Print the checks without running them") + parser.add_argument("--install", action="store_true", help="Install TypeScript dependencies first") + parser.add_argument("--live", action="store_true", help="Run the opt-in tests against Mem0 Platform") + parser.add_argument("--artifacts-dir", type=Path) + parser.add_argument("--report", type=Path) + args = parser.parse_args() + + selected = set(GROUPS if not args.group or "all" in args.group else args.group) + if args.live: + selected.add(LIVE_GROUP) + if args.list: + report = {"status": "planned", "checks": _planned_checks(selected)} + if args.report: + args.report.parent.mkdir(parents=True, exist_ok=True) + args.report.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + _print_report(report) + return 0 + if LIVE_GROUP in selected and not os.environ.get("MEM0_API_KEY"): + parser.error("MEM0_API_KEY is required for --live") + + temporary = None + if args.artifacts_dir: + artifacts_dir = args.artifacts_dir.resolve() + artifacts_dir.mkdir(parents=True, exist_ok=True) + else: + temporary = tempfile.TemporaryDirectory(prefix="mem0-plugin-conformance-") + artifacts_dir = Path(temporary.name) + + checks: list[dict[str, Any]] = [] + try: + if args.install: + checks.extend(_install_checks(selected)) + if "python-bundles" in selected: + checks.extend(_bundle_checks(artifacts_dir)) + checks.extend(_runtime_checks(selected)) + report = { + "status": "passed" if all(check["status"] == "passed" for check in checks) else "failed", + "checks": checks, + } + if args.report: + args.report.parent.mkdir(parents=True, exist_ok=True) + args.report.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + _print_report(report) + return 0 if report["status"] == "passed" else 1 + finally: + if temporary is not None: + temporary.cleanup() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/agent-plugin-core/python/flush_worker.py b/integrations/agent-plugin-core/python/flush_worker.py new file mode 100644 index 000000000..6f7b9ecb2 --- /dev/null +++ b/integrations/agent-plugin-core/python/flush_worker.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Detached remote checkpoint worker. + +Claude Code may cancel SessionEnd hooks as a print-mode process exits. The hook +therefore persists its input first and launches this process in a new session. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + checkpoint_session, + configure_harness, + touch_handoff_heartbeat, +) + + +def main() -> int: + if len(sys.argv) != 2: + return 2 + handoff_path = Path(sys.argv[1]) + os.environ["MEM0_CODE_HANDOFF_PATH"] = str(handoff_path) + harness = os.environ.get("MEM0_PLUGIN_HARNESS") + if harness: + source_tag = os.environ.get("MEM0_PLUGIN_SOURCE_TAG", "") + configure_harness( + harness, + env_prefix=os.environ.get("MEM0_PLUGIN_ENV_PREFIX", ""), + data_dir_name=os.environ.get("MEM0_PLUGIN_DATA_DIR_NAME", ""), + source_tag=source_tag, + ) + telemetry.init(harness=harness, source_tag=source_tag.upper()) + completed = False + try: + payload = json.loads(handoff_path.read_text(encoding="utf-8")) + delay = float(payload.get("delay_seconds") or 0) + if delay > 0: + payload.pop("delay_seconds", None) + temporary = handoff_path.with_suffix(f".{os.getpid()}.tmp") + try: + temporary.write_text(json.dumps(payload), encoding="utf-8") + temporary.replace(handoff_path) + finally: + temporary.unlink(missing_ok=True) + time.sleep(delay) + if not handoff_path.exists(): + return 0 + hook_input = payload.get("hook_input") or {} + reason = str(payload.get("reason") or "checkpoint") + wait_for_inflight = bool(payload.get("wait_for_inflight")) + store = EvidenceStore() + try: + if wait_for_inflight: + session_id = str( + hook_input.get("session_id") or "unknown-session" + ) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + deadline = time.monotonic() + float( + os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120") + ) + while ( + store.has_inflight_flush(repo.identity, session_id) + and time.monotonic() < deadline + ): + touch_handoff_heartbeat() + time.sleep(0.25) + # Hooks capture the conversation before handoff; the worker only flushes it. + result = checkpoint_session(store, hook_input, reason) + print(json.dumps(result, sort_keys=True), flush=True) + completed = result.get("status") in { + "semantic-succeeded", + "explicitly-stored", + "nothing-to-flush", + } + finally: + store.close() + return 0 + finally: + telemetry.flush() + if completed: + try: + handoff_path.unlink() + except OSError: + pass + elif handoff_path.suffix == ".running": + try: + handoff_path.replace(handoff_path.with_suffix(".json")) + except OSError: + pass + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/agent-plugin-core/python/hook_runner.py b/integrations/agent-plugin-core/python/hook_runner.py new file mode 100644 index 000000000..2a2adb557 --- /dev/null +++ b/integrations/agent-plugin-core/python/hook_runner.py @@ -0,0 +1,372 @@ +"""Shared hook orchestration for all Mem0 agent plugins.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import subprocess +import sys +import time +import uuid +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + _session_id, + api_key, + bounded, + cache_plugin_api_key, + checkpoint_session, + clear_stale_api_key_cache, + configure_harness, + data_dir, + detached_process_kwargs, + format_context, + harness_config, + record_session_start, + record_tool, + record_user_prompt, + redact, + search_memories, +) + +STALE_RUNNING_SECONDS = 300 +PENDING_EXPIRY_SECONDS = 7 * 24 * 60 * 60 +PENDING_LAUNCH_LIMIT = 5 +DEFAULT_IDLE_FLUSH_SECONDS = 300 + +_core_dir: Path = Path(__file__).resolve().parent + + +def read_hook_input() -> dict: + try: + value = json.load(sys.stdin) + return value if isinstance(value, dict) else {} + except (json.JSONDecodeError, OSError): + return {} + + +def default_record_stop(store: EvidenceStore, hook_input: dict): + """Record the assistant's response without transcript parsing.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + message = redact(hook_input.get("last_assistant_message", "")).strip() + if message: + store.record_assistant_response(repo, session_id, message) + return repo, session_id + + +def first_prompt_memory_output(store: EvidenceStore, hook_input: dict) -> dict: + """Search once before the agent handles the first prompt in a session.""" + repo, session_id, prompt, is_first_prompt = record_user_prompt(store, hook_input) + if not is_first_prompt: + return {} + try: + minimum_query_chars = int(os.environ.get("MEM0_CODE_MIN_QUERY_CHARS", "20")) + except ValueError: + minimum_query_chars = 20 + if len(prompt.strip()) < max(minimum_query_chars, 1): + return {} + result = search_memories( + store, repo, session_id, bounded(prompt, 6000), + top_k=5, operation="first-prompt-search", timeout=2, + ) + if not result.memories: + return {} + context = format_context( + result.memories, + "Mem0 found these relevant memories from earlier work in this repository:", + ) + telemetry.record( + "context_injected", + repo=repo, session_id=session_id, trigger="first-prompt", + memory_count=len(result.memories), context_chars=len(context), + prompt_chars=len(prompt), + ) + return { + "hookSpecificOutput": { + "hookEventName": "UserPromptSubmit", + "additionalContext": context, + }, + } + + +def _launch_handoff(handoff_path: Path) -> bool: + running_path = handoff_path.with_suffix(".running") + try: + handoff_path.replace(running_path) + except OSError: + return False + worker = _core_dir / "flush_worker.py" + log_path = data_dir() / "flush-worker.log" + log_handle = open(log_path, "a", encoding="utf-8") + harness = harness_config() + child_env = os.environ.copy() + child_env.update( + { + "MEM0_CODE_DATA_DIR": str(data_dir()), + "MEM0_PLUGIN_HARNESS": harness["name"], + "MEM0_PLUGIN_ENV_PREFIX": harness["env_prefix"], + "MEM0_PLUGIN_DATA_DIR_NAME": harness["data_dir_name"], + "MEM0_PLUGIN_SOURCE_TAG": harness["source_tag"], + } + ) + try: + subprocess.Popen( + [sys.executable, str(worker), str(running_path)], + stdin=subprocess.DEVNULL, + stdout=log_handle, stderr=log_handle, + close_fds=True, + env=child_env, + **detached_process_kwargs(), + ) + finally: + log_handle.close() + return True + + +def recover_pending_handoffs() -> int: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + now = time.time() + for running in pending_dir.glob("*.running"): + try: + if now - running.stat().st_mtime > STALE_RUNNING_SECONDS: + running.replace(running.with_suffix(".json")) + except OSError: + continue + recoverable = [] + for handoff in pending_dir.glob("*.json"): + try: + age = now - handoff.stat().st_mtime + except OSError: + continue + if age > PENDING_EXPIRY_SECONDS: + handoff.unlink(missing_ok=True) + continue + recoverable.append((age, handoff)) + recoverable.sort(key=lambda item: item[0], reverse=True) + launched = 0 + for _, handoff in recoverable[:PENDING_LAUNCH_LIMIT]: + launched += int(_launch_handoff(handoff)) + return launched + + +def refresh_pending_handoffs() -> None: + pending_dir = data_dir() / "pending" + if not pending_dir.is_dir(): + return + for pattern in ("*.json", "*.running"): + for handoff in pending_dir.glob(pattern): + try: + os.utime(handoff) + except OSError: + continue + + +def hand_off_flush( + hook_input: dict, reason: str, *, wait_for_inflight: bool = False, +) -> None: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = ( + f"{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}\0{reason}" + ) + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + handoff_path = pending_dir / f"{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": reason, + "wait_for_inflight": wait_for_inflight, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + + +def automatic_flush_enabled() -> bool: + return os.environ.get("MEM0_CODE_AUTO_FLUSH", "true").lower() in { + "1", "true", "yes", "on", + } + + +def schedule_periodic_checkpoint( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + if ( + not automatic_flush_enabled() + or not api_key() + or not store.checkpoint_due(repo.identity, session_id) + ): + return False + if store.prepare_flush(repo, session_id, "periodic") is None: + return False + hand_off_flush(hook_input, "periodic") + return True + + +def _idle_flush_seconds() -> int: + try: + return max( + int(os.environ.get("MEM0_CODE_IDLE_FLUSH_SECONDS", str(DEFAULT_IDLE_FLUSH_SECONDS))), + 0, + ) + except ValueError: + return DEFAULT_IDLE_FLUSH_SECONDS + + +def schedule_idle_flush( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + delay = _idle_flush_seconds() + if delay <= 0 or not automatic_flush_enabled() or not api_key(): + return False + if store.has_inflight_flush(repo.identity, session_id): + return False + if not store.has_unflushed_events(repo.identity, session_id): + return False + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = f"idle\0{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}" + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + for old in pending_dir.glob(f"idle-{digest}*"): + old.unlink(missing_ok=True) + handoff_path = pending_dir / f"idle-{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": "idle", + "delay_seconds": delay, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + return True + + +def log_failure(exc: Exception) -> None: + try: + log_path = data_dir() / "plugin-errors.log" + with log_path.open("a", encoding="utf-8") as handle: + handle.write(f"{time.time():.3f} {type(exc).__name__}: {exc}\n") + except OSError: + pass + + +def run( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> int: + if record_stop_fn is None: + record_stop_fn = default_record_stop + if automatic_flush_reasons is None: + automatic_flush_reasons = {"session-end"} + + base_actions = ["session-start", "user-prompt", "post-tool", "stop", "flush"] + all_actions = base_actions + list((extra_actions or {}).keys()) + + parser = argparse.ArgumentParser() + parser.add_argument("action", choices=all_actions) + parser.add_argument("--reason", default="manual") + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + args = parser.parse_args() + + if args.harness: + configure_harness(args.harness) + telemetry.init(harness=args.harness) + + if args.plugin_data_dir: + os.environ[data_dir_env] = args.plugin_data_dir + + cache_plugin_api_key() + if args.action == "session-start": + clear_stale_api_key_cache() + + hook_input = read_hook_input() + store = EvidenceStore() + try: + if store.is_paused(): + if args.action == "session-start": + refresh_pending_handoffs() + telemetry.record("session_start", paused=True) + telemetry.spawn_flush() + return 0 + + if args.action == "session-start": + if telemetry.is_first_run(): + telemetry.record("install") + recovered = recover_pending_handoffs() + record_session_start(store, hook_input) + if recovered: + telemetry.record("handoff_recovered", count=recovered) + telemetry.spawn_flush() + elif args.action == "user-prompt": + output = first_prompt_memory_output(store, hook_input) + if output: + print(json.dumps(output)) + elif args.action == "post-tool": + record_tool(store, hook_input) + elif args.action == "stop": + repo, session_id = record_stop_fn(store, hook_input) + if not schedule_periodic_checkpoint(store, hook_input, repo, session_id): + schedule_idle_flush(store, hook_input, repo, session_id) + elif args.action == "flush": + automatic = args.reason in automatic_flush_reasons + if automatic and not automatic_flush_enabled(): + return 0 + if args.reason == "session-end": + record_stop_fn(store, hook_input) + if os.environ.get("MEM0_CODE_SYNC_FLUSH") == "1": + print(json.dumps(checkpoint_session(store, hook_input, args.reason))) + else: + session_id = str(hook_input.get("session_id") or "unknown-session") + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + already_running = store.has_inflight_flush(repo.identity, session_id) + if already_running and args.reason == "session-end": + hand_off_flush(hook_input, args.reason, wait_for_inflight=True) + elif not already_running and store.prepare_flush( + repo, session_id, args.reason, + ) is not None: + hand_off_flush(hook_input, args.reason) + elif extra_actions and args.action in extra_actions: + result = extra_actions[args.action](store, hook_input) + if result: + print(json.dumps(result)) + finally: + store.close() + return 0 + + +def entry_point( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> None: + try: + raise SystemExit(run( + record_stop_fn=record_stop_fn, + extra_actions=extra_actions, + data_dir_env=data_dir_env, + automatic_flush_reasons=automatic_flush_reasons, + )) + except Exception as exc: + log_failure(exc) + raise SystemExit(0) + + +if __name__ == "__main__": + entry_point() diff --git a/integrations/agent-plugin-core/python/mcp_server.py b/integrations/agent-plugin-core/python/mcp_server.py new file mode 100644 index 000000000..036fbbdc9 --- /dev/null +++ b/integrations/agent-plugin-core/python/mcp_server.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Expose Mem0's memory search as one local coding-agent tool.""" + +from __future__ import annotations + +import json +import os +import sys +from typing import Any + +import telemetry +from memory_core import ( + CODING_MEMORY_CATEGORY_NAMES, + PLUGIN_VERSION, + SEARCH_SCOPES, + format_search_result, + resolve_repo, + search_memories, +) + +PROTOCOL_VERSION = "2024-11-05" +TOOL_NAME = "search_memories" +TOOL_DESCRIPTION = ( + "Search memories from earlier work in this repository. ALWAYS call this " + "tool before answering anything that could depend on prior context: the " + "user's preferences, facts about this codebase, history, people, projects, " + "or earlier decisions. Do not rely on the chat window alone. The " + "repository's memory is shared by everyone who works in it and includes " + "what it took to run, test, or build here, so search before assuming an " + "invocation works. The scope argument changes what is searched: 'repo' " + "(default) is the whole repository's shared memory plus your own " + "preferences, 'dir' narrows the shared part to the directory you are " + "working in, and 'mine' is your preferences alone." +) +TOOL_SCHEMA = { + "type": "object", + "properties": { + "query": { + "type": "string", + "minLength": 1, + "maxLength": 2000, + "description": "A direct question about earlier work in this repository.", + }, + "top_k": { + "type": "integer", + "minimum": 1, + "maximum": 20, + "description": "Maximum memories to return. Uses Mem0's configured default when omitted.", + }, + "category": { + "type": "string", + "enum": list(CODING_MEMORY_CATEGORY_NAMES), + "description": "Optional memory category. Omit to search every category.", + }, + "scope": { + "type": "string", + "enum": list(SEARCH_SCOPES), + "description": ( + "Which memories to search. 'repo' (default) is the whole repository's " + "shared memory plus your own preferences, 'dir' narrows the shared " + "part to the current directory, 'mine' is your preferences alone." + ), + }, + "run_id": { + "type": "string", + "minLength": 1, + "description": ( + "Optional coding-agent session ID. With any scope, restricts results to memories " + "saved in that session. Omit to recall memories across sessions." + ), + }, + }, + "required": ["query"], + "additionalProperties": False, +} + + +class ToolInputError(ValueError): + pass + + +def _validate_arguments( + arguments: Any, +) -> tuple[str, int | None, str | None, str | None, str | None]: + if not isinstance(arguments, dict): + raise ToolInputError("Search arguments must be an object.") + + unknown = set(arguments) - {"query", "top_k", "category", "scope", "run_id"} + if unknown: + raise ToolInputError(f"Unknown search argument: {sorted(unknown)[0]}") + + query = arguments.get("query") + if not isinstance(query, str) or not query.strip(): + raise ToolInputError("query must be a non-empty string.") + query = query.strip() + if len(query) > 2000: + raise ToolInputError("query must be at most 2,000 characters.") + + top_k = arguments.get("top_k") + if top_k is not None and ( + isinstance(top_k, bool) or not isinstance(top_k, int) or not 1 <= top_k <= 20 + ): + raise ToolInputError("top_k must be an integer from 1 to 20.") + + category = arguments.get("category") + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ToolInputError("category must be one of Mem0's supported categories.") + + scope = arguments.get("scope") + if scope is not None and scope not in SEARCH_SCOPES: + raise ToolInputError(f"scope must be one of {list(SEARCH_SCOPES)}.") + + run_id = arguments.get("run_id") + if run_id is not None: + if not isinstance(run_id, str) or not run_id.strip(): + raise ToolInputError("run_id must be a non-empty string.") + run_id = run_id.strip() + + return query, top_k, category, scope, run_id + + +def call_search_memories(arguments: Any, cwd: str | None = None) -> str: + query, top_k, category, scope, run_id = _validate_arguments(arguments) + repo = resolve_repo(cwd or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + result = search_memories( + None, + repo, + None, + query, + top_k=top_k, + category=category, + scope=scope, + run_id=run_id, + operation="mcp-search", + ) + return format_search_result(result) + + +def _workspace_cwd(params: dict[str, Any]) -> str | None: + meta = params.get("_meta") + if not isinstance(meta, dict): + return None + metadata = meta.get("x-codex-turn-metadata") + if not isinstance(metadata, dict): + return None + workspaces = metadata.get("workspaces") or {} + if isinstance(workspaces, dict): + return next((path for path in workspaces if isinstance(path, str) and path), None) + return None + + +def _tool_response(text: str, *, is_error: bool = False) -> dict[str, Any]: + return { + "content": [{"type": "text", "text": text}], + "isError": is_error, + } + + +def handle_request(message: Any) -> dict[str, Any] | None: + if not isinstance(message, dict): + return None + request_id = message.get("id") + method = message.get("method") + + if method == "notifications/initialized": + return None + if method == "initialize": + requested = (message.get("params") or {}).get("protocolVersion") + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "protocolVersion": requested or PROTOCOL_VERSION, + "capabilities": {"tools": {"listChanged": False}}, + "serverInfo": {"name": "mem0", "version": PLUGIN_VERSION}, + }, + } + if method == "ping": + return {"jsonrpc": "2.0", "id": request_id, "result": {}} + if method == "tools/list": + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "tools": [ + { + "name": TOOL_NAME, + "description": TOOL_DESCRIPTION, + "inputSchema": TOOL_SCHEMA, + "annotations": { + "readOnlyHint": True, + "idempotentHint": True, + "openWorldHint": True, + }, + } + ] + }, + } + if method == "tools/call": + params = message.get("params") or {} + if params.get("name") != TOOL_NAME: + result = _tool_response("Unknown Mem0 tool.", is_error=True) + else: + try: + result = _tool_response( + call_search_memories(params.get("arguments"), _workspace_cwd(params)) + ) + except ToolInputError as exc: + result = _tool_response(str(exc), is_error=True) + except Exception: + result = _tool_response("Memory search failed.", is_error=True) + return {"jsonrpc": "2.0", "id": request_id, "result": result} + if request_id is None: + return None + return { + "jsonrpc": "2.0", + "id": request_id, + "error": {"code": -32601, "message": "Method not found"}, + } + + +def main() -> int: + for raw_line in sys.stdin: + try: + message = json.loads(raw_line) + response = handle_request(message) + except json.JSONDecodeError: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32700, "message": "Parse error"}, + } + except Exception: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32603, "message": "Internal error"}, + } + if response is not None: + sys.stdout.write(json.dumps(response, separators=(",", ":")) + "\n") + sys.stdout.flush() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/agent-plugin-core/python/memory_cli.py b/integrations/agent-plugin-core/python/memory_cli.py new file mode 100644 index 000000000..595729feb --- /dev/null +++ b/integrations/agent-plugin-core/python/memory_cli.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Mem0 diagnostics and user controls.""" + +from __future__ import annotations + +import argparse +import json +import os + +import telemetry +from memory_core import ( + EvidenceStore, + api_key, + data_dir, + doctor, + forget_remote_repo, + configure_harness, + resolve_repo, + user_id, +) + + +def _print_status(value: dict) -> None: + last = value.get("last_operation") or {} + print(f"Mem0: {'paused' if value['paused'] else 'active'}") + print(f"Repository: {value['repo_id']}") + print(f"Local data: {value['data_dir']}") + print(f"API key: {'configured' if value['api_key_configured'] else 'missing'}") + print( + "Saved on this computer: " + f"{value['events']} session details, {value['flushes']} memory updates" + ) + print( + f"Used in this repository: {value['retrievals']} memories returned, " + f"{value['sidekick_runs']} sidekick runs" + ) + if last: + item_label = "" + if last["operation"] in {"flush", "flush-retry"}: + item_label = f", {last['item_count']} memories" + operation = ( + "memory update" + if last["operation"] in {"flush", "flush-retry"} + else last["operation"].replace("-", " ") + ) + print( + f"Last {operation}: " + f"{'succeeded' if last['success'] else 'failed'} " + f"({last['duration_ms']:.1f} ms{item_label})" + ) + sidekick = value.get("last_sidekick") or {} + if sidekick: + state = "finished" if sidekick.get("stopped_at") else "started" + print( + "Last sidekick: " + f"{state}, received {sidekick['context_chars']} characters of memory, " + f"agent {sidekick['agent_id']}" + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + subparsers = parser.add_subparsers(dest="command", required=True) + + status = subparsers.add_parser("status") + status.add_argument("--json", action="store_true") + + doctor_parser = subparsers.add_parser("doctor") + doctor_parser.add_argument("--json", action="store_true") + + subparsers.add_parser("pause") + subparsers.add_parser("resume") + + forget = subparsers.add_parser("forget") + forget.add_argument("--remote", action="store_true") + forget.add_argument("--yes", action="store_true") + forget.add_argument("--include-project-memory", action="store_true") + + args = parser.parse_args() + if args.harness: + source_tag = f"{args.harness.replace('-', '_')}_plugin" + configure_harness(args.harness, source_tag=source_tag) + telemetry.init(harness=args.harness, source_tag=source_tag.upper()) + if args.plugin_data_dir: + os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir + store = EvidenceStore() + try: + repo = resolve_repo(os.getcwd()) + telemetry.record("control", repo=repo, action=args.command) + if args.command == "status": + result = { + **store.status(repo.identity), + "repo_id": repo.identity, + "app_id": repo.app_id, + "project_id": repo.project_id, + "directory": repo.directory, + "user_id": user_id(), + "data_dir": str(data_dir()), + "api_key_configured": bool(api_key()), + } + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + _print_status(result) + elif args.command == "doctor": + result = doctor(os.getcwd()) + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + for name, check in result["checks"].items(): + print( + f"{'PASS' if check['ok'] else 'FAIL'} {name}: {check['detail']}" + ) + return 0 if result["ok"] else 1 + elif args.command == "pause": + store.set_setting("paused", "true") + print("Mem0 stopped saving and searching memories.") + elif args.command == "resume": + store.set_setting("paused", "false") + print("Mem0 resumed saving and searching memories.") + elif args.command == "forget": + if not args.yes: + print( + "Refusing to delete data without --yes. Add --remote to also " + "delete this user/repository scope from Mem0." + ) + return 2 + remote_result = ( + forget_remote_repo( + repo, include_project_memory=args.include_project_memory + ) + if args.remote + else None + ) + local_result = store.forget_local_repo(repo.identity) + print( + json.dumps( + {"local": local_result, "remote": remote_result}, + indent=2, + default=str, + ) + ) + if remote_result and remote_result.get("status") == "error": + return 1 + finally: + store.close() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/agent-plugin-core/python/memory_core.py b/integrations/agent-plugin-core/python/memory_core.py new file mode 100644 index 000000000..cf71196b8 --- /dev/null +++ b/integrations/agent-plugin-core/python/memory_core.py @@ -0,0 +1,2645 @@ +#!/usr/bin/env python3 +"""Shared core for Mem0 agent plugins. + +Hooks record small session details locally. When the agent compacts or ends the +session, Mem0 sends the useful parts to the platform so it can create memories. +The agent can search those memories during later work in the repository. +""" + +from __future__ import annotations + +import functools +import hashlib +import json +import math +import os +import re +import sqlite3 +import subprocess +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +import telemetry + +DEFAULT_API_URL = "https://api.mem0.ai" +PLUGIN_VERSION = "0.3.1" + +_harness_name: str = "generic" +_harness_env_prefix: str = "MEM0_PLUGIN" +_harness_data_dir_name: str = "mem0-plugin" +_harness_source_tag: str = "mem0_plugin" + + +def configure_harness( + name: str, + env_prefix: str = "", + data_dir_name: str = "", + source_tag: str = "", +) -> None: + global _harness_name, _harness_env_prefix, _harness_data_dir_name, _harness_source_tag + _harness_name = name + _harness_env_prefix = env_prefix or f"MEM0_{name.upper().replace('-', '_')}" + _harness_data_dir_name = data_dir_name or f"{name}-plugin" + _harness_source_tag = source_tag or f"{name.replace('-', '_')}_plugin" + + +def harness_config() -> dict[str, str]: + return { + "name": _harness_name, + "env_prefix": _harness_env_prefix, + "data_dir_name": _harness_data_dir_name, + "source_tag": _harness_source_tag, + } + + +MAX_COMMAND_CHARS = 2000 +MAX_RESULT_CHARS = 2500 +MAX_EPISODE_CHARS = 12000 +CHECKPOINT_EXCHANGES = 5 +CHECKPOINT_MESSAGES = 10 +CHECKPOINT_SOURCE_CHARS = 40000 +DEFAULT_MAX_CONTEXT_CHARS = 4000 +MAX_EXTRACTION_INPUT_TOKENS = 24000 +MAX_FLUSH_ATTEMPTS = 5 +FORGET_PAGE_SIZE = 100 +FORGET_MAX_PAGES = 50 + +PROJECT_MEMORY_INSTRUCTIONS = """Save concise repository facts that will help anyone with future coding work in this repository. + +A completed change should produce one memory explaining the resulting behavior, where it is implemented when useful, and any important constraints or reasoning. Exploration or accepted decisions may produce separate memories only when they are independently useful. + +A command that failed and was then made to work should produce one memory naming the failing invocation, the error it returned, and the invocation that succeeded. Do not save one-off errors caused by an edit still in progress, transient network failures, or anything a rerun would fix on its own. + +Use the current coding agent's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or the coding agent completed them. Treat subagent responses as supporting repository evidence, not as decisions. + +Write about the repository, not the user, assistant, session, or task. Do not save personal preferences. Do not save a memory that only states which repository, branch, or directory the session worked in. Do not include test results, documentation updates, release notes, or temporary state. + +If nothing useful was established, return no memories.""" + +PERSONAL_MEMORY_INSTRUCTIONS = """Save concise facts about the user that will help in any repository: preferred tools, package managers, languages, coding style, review and communication preferences, and anything the user explicitly asked to be remembered about themselves. + +Write in the third person about the user, not about the repository, the assistant, the session, or the task. Do not save repository facts, project decisions, commands, or what was built. + +Never save that the user has no preferences or that nothing was learned. If nothing was learned about the user, return no memories.""" + +CODING_MEMORY_CATEGORIES = [ + { + "project_knowledge": ( + "What the project is and how its code, APIs, data, files, and " + "components work." + ) + }, + { + "decisions_and_constraints": ( + "Why an approach was chosen, what must remain true, and rules future " + "work must follow." + ) + }, + { + "workflows": ( + "How to run, test, debug, deploy, configure, or otherwise work on the " + "project." + ) + }, + { + "problems_and_fixes": ( + "Bugs, failures, known pitfalls, their causes, and how to fix or avoid " + "them." + ) + }, + { + "results": ( + "Outcomes and measurements from tests, benchmarks, experiments, or " + "investigations." + ) + }, +] +CODING_MEMORY_CATEGORY_NAMES = tuple( + category_name + for category in CODING_MEMORY_CATEGORIES + for category_name in category +) + +TEST_COMMAND_RE = re.compile( + r"(?:^|\s)(?:pytest|py\.test|jest|vitest|go\s+test|cargo\s+test|" + r"npm\s+(?:run\s+)?test|pnpm\s+(?:run\s+)?test|yarn\s+test|" + r"mvn\s+test|gradle\s+test|make\s+test)(?:\s|$)", + re.IGNORECASE, +) +BUILD_COMMAND_RE = re.compile( + r"(?:^|\s)(?:npm|pnpm|yarn)\s+(?:run\s+)?build(?:\s|$)|" + r"(?:^|\s)(?:cargo|go|mvn|gradle|make)\s+build(?:\s|$)", + re.IGNORECASE, +) + +SECRET_PATTERNS = [ + re.compile(r"(?i)(authorization\s*[:=]\s*(?:bearer|token)\s+)[^\s\"']+"), + re.compile( + r"(?i)((?:api[_-]?key|secret[_-]?access[_-]?key|session[_-]?token)\s*[:=]\s*)[^\s\"']+" + ), + re.compile( + r"(?i)((?:access[_-]?token|refresh[_-]?token|password|credential)" + r"\s*[:=]\s*)[^\s&\"']+" + ), + re.compile(r"\b(?:sk|m0|mem0_sk|psk)-[A-Za-z0-9_\-]{12,}\b"), + re.compile(r"\b(?:ASIA|AKIA)[A-Z0-9]{12,}\b"), + re.compile(r"\b(?:ghp_|github_pat_|xox[baprs]-)[A-Za-z0-9_\-]{12,}\b"), + re.compile( + r"-----BEGIN [^-]*PRIVATE KEY-----.*?-----END [^-]*PRIVATE KEY-----", + re.DOTALL, + ), + re.compile( + r'(?i)("(?:api[_-]?key|password|secret(?:[_-]?access[_-]?key)?' + r'|(?:access|refresh|session)[_-]?token|token|authorization|credential' + r')"\s*:\s*")(?:\\.|[^"\\])*' + ), +] + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def redact(value: Any) -> str: + text = ( + value + if isinstance(value, str) + else json.dumps(value, ensure_ascii=False, default=str) + ) + for pattern in SECRET_PATTERNS: + if pattern.groups: + text = pattern.sub(r"\1[REDACTED]", text) + else: + text = pattern.sub("[REDACTED]", text) + return text + + +def bounded(value: Any, limit: int) -> str: + text = redact(value).strip() + if len(text) <= limit: + return text + return text[:limit] + f"\n...[truncated {len(text) - limit} chars]" + + +def _git(cwd: str, *args: str) -> str: + try: + result = subprocess.run( + ["git", "-C", cwd, *args], + check=False, + capture_output=True, + text=True, + timeout=0.5, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return result.stdout.strip() if result.returncode == 0 else "" + + +def _normalize_remote(remote: str) -> str: + remote = remote.strip() + if remote.startswith("git@") and ":" in remote: + host_path = remote[4:].replace(":", "/", 1) + remote = f"https://{host_path}" + if remote.endswith(".git"): + remote = remote[:-4] + if "://" in remote: + parsed = urllib.parse.urlsplit(remote) + hostname = parsed.hostname or "" + if parsed.port: + hostname = f"{hostname}:{parsed.port}" + remote = urllib.parse.urlunsplit( + (parsed.scheme, hostname, parsed.path, parsed.query, parsed.fragment) + ) + return remote.rstrip("/") + + +_WILDCARD_SCOPE = re.compile(r"^\*+$") + + +def _scope_value(raw: str | None) -> str: + """Reject wildcards as identities: they are filter syntax and would widen the scope.""" + value = (raw or "").strip() + return "" if _WILDCARD_SCOPE.match(value) else value + + +SEARCH_SCOPES = ("repo", "dir", "mine") +DEFAULT_SEARCH_SCOPE = "repo" + +def directory_app_id(repo: RepoContext) -> str: + """The app_id of the directory this session runs in: the repository at the root, repository/path below it.""" + return f"{repo.app_id}/{repo.directory}" if repo.directory else repo.app_id + + +def directory_chain(repo: RepoContext) -> list[str]: + """Every directory a memory belongs to, from the top-level folder down to the one it was written in.""" + parts = repo.directory.split("/") if repo.directory else [] + return ["/".join(parts[: index + 1]) for index in range(len(parts))] + + +def _shared_project_ids(repo: RepoContext) -> list[str]: + """Current and pre-upgrade namespaces, shared by recall and explicit deletion.""" + if not repo.identity.startswith("local:") and repo.project_id != repo.app_id: + return [repo.project_id, repo.app_id] + return [repo.project_id] + + +def _search_filters(user: str, repo: RepoContext, scope: str) -> dict[str, Any]: + """Build the scope filter: app_id scopes to the repo, then union shared and personal lanes.""" + app_scope = {"app_id": repo.app_id} + mine = {"AND": [{"user_id": user}, app_scope]} + if scope == "mine": + return mine + projects = [{"AND": [{"agent_id": project_id}, app_scope]} for project_id in _shared_project_ids(repo)] + shared: dict[str, Any] = projects[0] if len(projects) == 1 else {"OR": projects} + if scope == "dir" and repo.directory: + shared = {"AND": [shared, {"metadata": {"dirs": {"contains": repo.directory}}}]} + return {"OR": [shared, mine]} + + +def search_scope() -> str: + configured = ( + _plugin_option("search_scope", "MEM0_CODE_SEARCH_SCOPE") or "" + ).strip().lower() + return configured if configured in SEARCH_SCOPES else DEFAULT_SEARCH_SCOPE + + +def resolve_search_scope(scope: str | None) -> str: + value = (scope or search_scope()).strip().lower() + if value not in SEARCH_SCOPES: + raise ValueError(f"Unknown search scope: {value}") + return value + + +def _legacy_project_map(cwd: str, root: str, raw_remote: str) -> str: + """Return the project name used by the previous Claude Code plugin.""" + try: + data = json.loads((Path.home() / ".mem0" / "project_map.json").read_text()) + except (OSError, json.JSONDecodeError): + return "" + if not isinstance(data, dict): + return "" + + keys = list(dict.fromkeys([cwd, root, os.path.realpath(cwd), os.path.realpath(root)])) + if raw_remote: + keys.append(f"remote:{hashlib.sha256(raw_remote.encode()).hexdigest()[:16]}") + for key in keys: + value = data.get(key) + if isinstance(value, str) and value.strip(): + return value.strip() + return "" + + +def _legacy_project_id(cwd: str, root: str, raw_remote: str, identity: str) -> str: + """Use the repository namespace created by the previous Mem0 plugin.""" + configured = _scope_value(os.environ.get("MEM0_PROJECT_ID")) + if configured: + return configured + + mapped = _scope_value(_legacy_project_map(cwd, root, raw_remote)) + if mapped: + return mapped + + remote = raw_remote or ("" if identity.startswith("local:") else identity) + remote = remote.strip().removesuffix(".git") + for prefix in ("https://", "http://", "ssh://", "git://"): + if remote.startswith(prefix): + remote = remote[len(prefix) :] + break + else: + remote = re.sub(r"^git@", "", remote) + parts = [part for part in remote.replace(":", "/", 1).split("/") if part] + if len(parts) >= 2: + return f"{parts[-2]}-{parts[-1]}".replace("/", "-").replace(":", "-") + if parts: + return parts[-1].replace("/", "-").replace(":", "-") + return os.path.basename(root or cwd) or "unknown" + + +@dataclass(frozen=True) +class RepoContext: + cwd: str + root: str + identity: str + app_id: str + branch: str + head_sha: str + project_id: str = "" + directory: str = "" + + +def _project_id(root: str, identity: str, app_id: str) -> str: + """The shared namespace: includes a host hash so repos with the same owner/name on different hosts stay apart.""" + if not identity.startswith("local:"): + return f"{app_id}-{hashlib.sha256(identity.encode()).hexdigest()[:10]}" + return f"local-{app_id}-{hashlib.sha256(root.encode()).hexdigest()[:10]}" + + +def _relative_directory(cwd: str, root: str) -> str: + relative = os.path.relpath(cwd, root) + return "" if relative == "." or relative.startswith("..") else relative.replace(os.sep, "/") + + +@dataclass(frozen=True) +class MemorySearchResult: + succeeded: bool + matched_count: int + already_shown_count: int + memories: list[dict[str, Any]] + + +@functools.lru_cache(maxsize=64) +def _resolve_repo_cached(cwd: str) -> RepoContext: + given_cwd = cwd + cwd = os.path.realpath(cwd) + given_root = _git(cwd, "rev-parse", "--show-toplevel") or given_cwd + root = os.path.realpath(given_root) + raw_remote = _git(root, "config", "--get", "remote.origin.url") + remote = _normalize_remote(raw_remote) + identity = remote or f"local:{root}" + app_id = _legacy_project_id(given_cwd, given_root, raw_remote, identity) + return RepoContext( + cwd=cwd, + root=root, + identity=identity, + app_id=app_id, + branch=_git(root, "branch", "--show-current") or "detached", + head_sha=_git(root, "rev-parse", "HEAD"), + project_id=_project_id(root, identity, app_id), + directory=_relative_directory(cwd, root), + ) + + +def resolve_repo(cwd: str | None) -> RepoContext: + return _resolve_repo_cached(os.path.abspath(cwd or os.getcwd())) + + +def api_key() -> str: + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return configured + try: + return (data_dir() / "api-key").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def cache_plugin_api_key() -> bool: + """Bridge host's hook-only sensitive config into plugin-owned storage.""" + configured = ( + os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if not configured: + return False + + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + path = directory / "api-key" + temporary = directory / f"api-key.{os.getpid()}.tmp" + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_TRUNC, + 0o600, + ) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + handle.write(configured) + os.replace(temporary, path) + os.chmod(path, 0o600) + finally: + try: + temporary.unlink() + except FileNotFoundError: + pass + return True + + +def clear_stale_api_key_cache() -> bool: + """Drop the cached key file once every configured key source is gone.""" + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return False + path = data_dir() / "api-key" + if not path.exists(): + return False + try: + path.unlink() + except OSError: + return False + return True + + +def detached_process_kwargs(platform: str | None = None) -> dict: + """Keep a spawned worker alive after the coding agent exits, on POSIX and Windows.""" + if (platform or sys.platform) == "win32": + return { + "creationflags": subprocess.DETACHED_PROCESS + | subprocess.CREATE_NEW_PROCESS_GROUP + } + return {"start_new_session": True} + + +def _plugin_option(name: str, fallback: str = "") -> str: + return ( + os.environ.get(f"PLUGIN_OPTION_{name.upper()}") + or os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") + or os.environ.get(fallback) + or "" + ).strip() + + +def user_id() -> str: + return ( + _scope_value(_plugin_option("user_id", "MEM0_CODE_USER_ID")) + or _scope_value(os.environ.get("MEM0_USER_ID")) + or _scope_value(os.environ.get("MEM0_RESOLVED_USER_ID")) + or _scope_value(os.environ.get("USER")) + or _scope_value(os.environ.get("USERNAME")) + or "default" + ) + + +def data_dir() -> Path: + configured = ( + os.environ.get("MEM0_CODE_DATA_DIR") + or os.environ.get("MEM0_PLUGIN_DATA_DIR") + or os.environ.get("PLUGIN_DATA") + or os.environ.get("CLAUDE_PLUGIN_DATA") + ) + return ( + Path(configured).expanduser() if configured else Path.home() / ".mem0" / _harness_data_dir_name + ) + + +def _bool_option(name: str, fallback: str, default: bool = False) -> bool: + value = _plugin_option(name, fallback) + if not value: + return default + return value.lower() in {"1", "true", "yes", "on"} + + +def _int_option(name: str, fallback: str, default: int) -> int: + value = _plugin_option(name, fallback) + try: + return int(value) if value else default + except ValueError: + return default + + + +def _checkpoint_message(event: dict[str, Any]) -> str: + kind = event.get("kind") + payload = event.get("payload") or {} + if kind == "user_prompt": + return redact(payload.get("text", "")).strip() + if kind == "assistant_stop": + transcript_messages = payload.get("transcript_messages") or [] + if isinstance(transcript_messages, list): + text = "\n".join( + str(message.get("content") or "") + for message in transcript_messages + if isinstance(message, dict) and message.get("content") + ) + if text: + return text + return redact(payload.get("text", "")).strip() + if kind == "sidekick_stop": + return redact(payload.get("final_message", "")).strip() + return "" + + +def checkpoint_stats(events: list[dict[str, Any]]) -> tuple[int, int, int]: + """Return completed exchanges, messages, and source characters.""" + completed = sum(event.get("kind") == "assistant_stop" for event in events) + contents = [content for event in events if (content := _checkpoint_message(event))] + return completed, len(contents), sum(len(content) for content in contents) + + +def select_checkpoint_events( + events: list[dict[str, Any]], *, force: bool +) -> list[dict[str, Any]]: + """Select one ordered extraction block without splitting an exchange.""" + for index, event in enumerate(events): + if event.get("kind") != "assistant_stop": + continue + candidate = events[: index + 1] + completed, messages, source_chars = checkpoint_stats(candidate) + if ( + completed >= CHECKPOINT_EXCHANGES + or messages >= CHECKPOINT_MESSAGES + or source_chars >= CHECKPOINT_SOURCE_CHARS + ): + return candidate + return events if force else [] + + +class EvidenceStore: + def __init__(self, path: Path | None = None): + directory = data_dir() if path is None else path.parent + directory.mkdir(parents=True, exist_ok=True) + self.path = path or directory / "evidence.sqlite3" + try: + self._open() + except sqlite3.DatabaseError: + self._quarantine() + self._open() + + def _open(self) -> None: + self.conn = sqlite3.connect(self.path, timeout=10) + self.conn.row_factory = sqlite3.Row + try: + self.conn.execute("PRAGMA journal_mode=WAL") + self.conn.execute("PRAGMA busy_timeout=10000") + self._migrate() + except sqlite3.DatabaseError: + self.conn.close() + raise + + def _quarantine(self) -> None: + """Move an unreadable database aside so capture restarts cleanly.""" + stamp = int(time.time()) + for suffix in ("", "-wal", "-shm"): + source = Path(f"{self.path}{suffix}") + try: + source.replace(f"{self.path}.corrupt-{stamp}{suffix}") + except FileNotFoundError: + continue + except OSError: + try: + source.unlink() + except OSError: + pass + telemetry.record("db_quarantined") + + def close(self) -> None: + self.conn.close() + + def _migrate(self) -> None: + self.conn.executescript( + """ + CREATE TABLE IF NOT EXISTS events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + created_at TEXT NOT NULL, + kind TEXT NOT NULL, + payload_json TEXT NOT NULL, + flush_id TEXT + ); + CREATE INDEX IF NOT EXISTS events_session_idx + ON events(repo_id, session_id, flush_id, id); + + CREATE TABLE IF NOT EXISTS session_scopes ( + session_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + root TEXT NOT NULL, + branch TEXT NOT NULL, + head_sha TEXT NOT NULL, + created_at TEXT NOT NULL, + directory TEXT NOT NULL DEFAULT '' + ); + + CREATE TABLE IF NOT EXISTS flushes ( + packet_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + reason TEXT NOT NULL, + event_start INTEGER NOT NULL, + event_end INTEGER NOT NULL, + status TEXT NOT NULL, + episode_event_id TEXT, + semantic_event_id TEXT, + error TEXT, + attempts INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + CREATE TABLE IF NOT EXISTS retrievals ( + session_id TEXT NOT NULL, + repo_id TEXT NOT NULL, + memory_id TEXT NOT NULL, + injected_at TEXT NOT NULL, + rank INTEGER, + score REAL, + memory_text TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(session_id, repo_id, memory_id) + ); + + CREATE TABLE IF NOT EXISTS operations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + created_at TEXT NOT NULL, + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + operation TEXT NOT NULL, + duration_ms REAL NOT NULL, + success INTEGER NOT NULL, + item_count INTEGER NOT NULL DEFAULT 0, + request_chars INTEGER NOT NULL DEFAULT 0, + response_chars INTEGER NOT NULL DEFAULT 0, + error TEXT + ); + + CREATE TABLE IF NOT EXISTS sidekick_runs ( + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + agent_type TEXT NOT NULL, + started_at TEXT NOT NULL, + stopped_at TEXT, + transcript_path TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + final_message TEXT, + PRIMARY KEY(repo_id, session_id, agent_id) + ); + CREATE INDEX IF NOT EXISTS sidekick_runs_repo_idx + ON sidekick_runs(repo_id, started_at); + + CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + """ + ) + # Remove the pre-0.1.1 no-tools snapshot implementation. The real coding + # sidekick is a native Claude Code agent and stores no state in this DB. + self.conn.executescript( + """ + DROP TABLE IF EXISTS sidekick_calls; + DROP TABLE IF EXISTS sidekick_snapshots; + DROP TABLE IF EXISTS sidekick_state; + DROP TABLE IF EXISTS sidekick_packets; + """ + ) + self._ensure_column("retrievals", "rank", "INTEGER") + self._ensure_column("retrievals", "score", "REAL") + self._ensure_column("retrievals", "memory_text", "TEXT") + self._ensure_column("retrievals", "context_chars", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("flushes", "attempts", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("session_scopes", "directory", "TEXT NOT NULL DEFAULT ''") + self.conn.commit() + + def _ensure_column(self, table: str, column: str, declaration: str) -> None: + columns = { + str(row["name"]) + for row in self.conn.execute(f"PRAGMA table_info({table})").fetchall() + } + if column not in columns: + self.conn.execute(f"ALTER TABLE {table} ADD COLUMN {column} {declaration}") + + def record_event( + self, + repo: RepoContext, + session_id: str, + kind: str, + payload: dict[str, Any], + ) -> int: + cursor = self.conn.execute( + """INSERT INTO events + (repo_id, app_id, session_id, created_at, kind, payload_json) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + repo.app_id, + session_id, + utc_now(), + kind, + json.dumps(payload, ensure_ascii=False, sort_keys=True), + ), + ) + self.conn.commit() + return int(cursor.lastrowid) + + def record_assistant_response(self, repo: RepoContext, session_id: str, message: str) -> None: + """Ignore repeated response hooks until another prompt or a different answer arrives.""" + with self.conn: + # Serialize the check and insert across concurrent Stop and SessionEnd hooks. + self.conn.execute("BEGIN IMMEDIATE") + previous = self.conn.execute( + """SELECT kind, payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind IN ('user_prompt', 'assistant_stop') + ORDER BY id DESC LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if ( + previous is not None + and previous["kind"] == "assistant_stop" + and json.loads(previous["payload_json"]).get("text") == message + ): + return + self.record_event(repo, session_id, "assistant_stop", {"text": message}) + + def repo_for_session(self, session_id: str, cwd: str | None) -> RepoContext: + """Keep one project scope for every hook in a coding-agent session.""" + current = resolve_repo(cwd) + if session_id == "unknown-session": + return current + + with self.conn: + self.conn.execute( + """INSERT OR IGNORE INTO session_scopes + (session_id, repo_id, app_id, root, branch, head_sha, created_at, directory) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + current.identity, + current.app_id, + current.root, + current.branch, + current.head_sha, + utc_now(), + current.directory, + ), + ) + scope = self.conn.execute( + "SELECT * FROM session_scopes WHERE session_id = ?", (session_id,) + ).fetchone() + same_git_repo = ( + current.identity == scope["repo_id"] and bool(current.head_sha) + ) + pinned = current if same_git_repo else resolve_repo(str(scope["root"])) + return RepoContext( + cwd=current.cwd, + root=pinned.root, + identity=str(scope["repo_id"]), + app_id=str(scope["app_id"]), + branch=pinned.branch, + head_sha=pinned.head_sha, + project_id=pinned.project_id, + directory=str(scope["directory"] or ""), + ) + + def prepare_flush( + self, repo: RepoContext, session_id: str, reason: str + ) -> tuple[str, list[dict[str, Any]]] | None: + with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + existing = self.conn.execute( + """SELECT * FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') + ORDER BY created_at LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: + self.conn.execute( + "UPDATE flushes SET status = 'gave-up', updated_at = ? WHERE packet_id = ?", + (utc_now(), existing["packet_id"]), + ) + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status="gave-up", + success=False, + attempts=int(existing["attempts"] or 0), + ) + existing = None + if existing: + if reason != "periodic" and existing["reason"] == "periodic": + self.conn.execute( + "UPDATE flushes SET reason = ?, updated_at = ? WHERE packet_id = ?", + (reason, utc_now(), existing["packet_id"]), + ) + existing_rows = self.conn.execute( + "SELECT * FROM events WHERE flush_id = ? ORDER BY id", + (existing["packet_id"],), + ).fetchall() + if existing_rows: + return str(existing["packet_id"]), [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in existing_rows + ] + + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo.identity, session_id), + ).fetchall() + if not rows: + return None + + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + events = select_checkpoint_events(events, force=reason != "periodic") + if not events: + return None + event_start, event_end = events[0]["id"], events[-1]["id"] + packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" + packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] + now = utc_now() + + self.conn.execute( + """INSERT OR IGNORE INTO flushes + (packet_id, repo_id, app_id, session_id, reason, event_start, + event_end, status, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, 'prepared', ?, ?)""", + ( + packet_id, + repo.identity, + repo.app_id, + session_id, + reason, + event_start, + event_end, + now, + now, + ), + ) + event_ids = [event["id"] for event in events] + placeholders = ", ".join("?" for _ in event_ids) + self.conn.execute( + f"UPDATE events SET flush_id = ? " + f"WHERE id IN ({placeholders}) AND flush_id IS NULL", + (packet_id, *event_ids), + ) + return packet_id, events + + def checkpoint_due(self, repo_id: str, session_id: str) -> bool: + if self.has_inflight_flush(repo_id, session_id): + return False + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo_id, session_id), + ).fetchall() + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + return bool(select_checkpoint_events(events, force=False)) + + def has_inflight_flush(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status IN ('prepared', 'semantic-queued') + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def flush_record(self, packet_id: str) -> dict[str, Any] | None: + row = self.conn.execute( + "SELECT * FROM flushes WHERE packet_id = ?", (packet_id,) + ).fetchone() + return dict(row) if row else None + + def has_unflushed_events(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def unflushed_starts_with_session_start( + self, repo_id: str, session_id: str + ) -> bool: + row = self.conn.execute( + """SELECT kind FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id LIMIT 1""", + (repo_id, session_id), + ).fetchone() + return bool(row and row["kind"] == "session_start") + + def update_flush(self, packet_id: str, **fields: Any) -> None: + allowed = {"status", "episode_event_id", "semantic_event_id", "error"} + updates = {key: value for key, value in fields.items() if key in allowed} + updates["updated_at"] = utc_now() + clause = ", ".join(f"{key} = ?" for key in updates) + failed = str(fields.get("status", "")) in { + "error", + "semantic-failed", + "semantic-timeout", + "semantic-missing", + } + if failed: + clause += ", attempts = attempts + 1" + with self.conn: + self.conn.execute( + f"UPDATE flushes SET {clause} WHERE packet_id = ?", + [*updates.values(), packet_id], + ) + + def unseen( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> list[dict[str, Any]]: + seen = { + row["memory_id"] + for row in self.conn.execute( + "SELECT memory_id FROM retrievals WHERE session_id = ? AND repo_id = ?", + (session_id, repo_id), + ) + } + return [memory for memory in memories if str(memory.get("id", "")) not in seen] + + def mark_injected( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> None: + now = utc_now() + with self.conn: + for rank, memory in enumerate(memories, start=1): + memory_id = str(memory.get("id", "")) + if memory_id: + memory_text = bounded( + memory.get("memory") or memory.get("text") or "", + 4000, + ) + try: + score = float(memory["score"]) + except (KeyError, TypeError, ValueError): + score = None + self.conn.execute( + """INSERT OR IGNORE INTO retrievals + (session_id, repo_id, memory_id, injected_at, rank, + score, memory_text, context_chars) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + repo_id, + memory_id, + now, + rank, + score, + memory_text, + len(memory_text), + ), + ) + + def injected_memories( + self, session_id: str, repo_id: str + ) -> list[dict[str, Any]]: + """Return the exact memories already supplied to the main conversation.""" + rows = self.conn.execute( + """SELECT memory_id, rank, score, memory_text + FROM retrievals + WHERE session_id = ? AND repo_id = ? + ORDER BY COALESCE(rank, 2147483647), injected_at, memory_id""", + (session_id, repo_id), + ).fetchall() + return [ + { + "id": row["memory_id"], + "memory": row["memory_text"], + "score": row["score"], + } + for row in rows + if row["memory_text"] + ] + + def start_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + context_chars: int, + ) -> bool: + """Record one native sidekick instance and whether context was first sent.""" + with self.conn: + cursor = self.conn.execute( + """INSERT OR IGNORE INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + context_chars) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + utc_now(), + context_chars, + ), + ) + return int(cursor.rowcount) > 0 + + def stop_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + transcript_path: str, + final_message: str, + ) -> str: + now = utc_now() + self.conn.execute("BEGIN IMMEDIATE") + try: + if not agent_id: + rows = self.conn.execute( + """SELECT agent_id FROM sidekick_runs + WHERE repo_id = ? AND session_id = ? AND agent_type = ? AND stopped_at IS NULL + LIMIT 2""", + (repo.identity, session_id, agent_type), + ).fetchall() + # Without a host ID, overlapping runs cannot be correlated reliably. + agent_id = rows[0]["agent_id"] if len(rows) == 1 else f"unknown-agent-{time.time_ns()}" + self.conn.execute( + """INSERT INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + stopped_at, transcript_path, final_message) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(repo_id, session_id, agent_id) DO UPDATE SET + stopped_at = excluded.stopped_at, + transcript_path = excluded.transcript_path, + final_message = excluded.final_message""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + now, + now, + bounded(transcript_path, 2000), + redact(final_message).strip(), + ), + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + return agent_id + + def operation( + self, + repo: RepoContext, + session_id: str, + operation: str, + duration_ms: float, + success: bool, + *, + item_count: int = 0, + request_chars: int = 0, + response_chars: int = 0, + error: str = "", + ) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO operations + (created_at, repo_id, session_id, operation, duration_ms, + success, item_count, request_chars, response_chars, error) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + ( + utc_now(), + repo.identity, + session_id, + operation, + duration_ms, + int(success), + item_count, + request_chars, + response_chars, + bounded(error, 1000), + ), + ) + + def has_operation(self, repo_id: str, session_id: str, operation: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM operations + WHERE repo_id = ? AND session_id = ? AND operation = ? + LIMIT 1""", + (repo_id, session_id, operation), + ).fetchone() + return row is not None + + def has_event(self, repo_id: str, session_id: str, kind: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + return row is not None + + def latest_event_payload( + self, repo_id: str, session_id: str, kind: str + ) -> dict[str, Any]: + row = self.conn.execute( + """SELECT payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + ORDER BY id DESC LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + if not row: + return {} + try: + payload = json.loads(row["payload_json"]) + except json.JSONDecodeError: + return {} + return payload if isinstance(payload, dict) else {} + + def setting(self, key: str, default: str = "") -> str: + row = self.conn.execute( + "SELECT value FROM settings WHERE key = ?", (key,) + ).fetchone() + return str(row["value"]) if row else default + + def set_setting(self, key: str, value: str) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO settings(key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at""", + (key, value, utc_now()), + ) + + def is_paused(self) -> bool: + return self.setting("paused", "false").lower() in { + "1", + "true", + "yes", + "on", + } + + def forget_local_repo(self, repo_id: str) -> dict[str, int]: + tables = { + "events": "repo_id", + "session_scopes": "repo_id", + "flushes": "repo_id", + "retrievals": "repo_id", + "operations": "repo_id", + "sidekick_runs": "repo_id", + } + removed: dict[str, int] = {} + with self.conn: + for table, column in tables.items(): + cursor = self.conn.execute( + f"DELETE FROM {table} WHERE {column} = ?", (repo_id,) + ) + removed[table] = max(int(cursor.rowcount), 0) + return removed + + def status(self, repo_id: str) -> dict[str, Any]: + def count(table: str) -> int: + return int( + self.conn.execute( + f"SELECT COUNT(*) FROM {table} WHERE repo_id = ?", (repo_id,) + ).fetchone()[0] + ) + + last_operation = self.conn.execute( + """SELECT created_at, operation, duration_ms, success, item_count, error + FROM operations WHERE repo_id = ? ORDER BY id DESC LIMIT 1""", + (repo_id,), + ).fetchone() + last_sidekick = self.conn.execute( + """SELECT session_id, agent_id, agent_type, started_at, stopped_at, + context_chars + FROM sidekick_runs WHERE repo_id = ? + ORDER BY started_at DESC LIMIT 1""", + (repo_id,), + ).fetchone() + return { + "paused": self.is_paused(), + "events": count("events"), + "flushes": count("flushes"), + "retrievals": count("retrievals"), + "sidekick_runs": count("sidekick_runs"), + "last_operation": dict(last_operation) if last_operation else None, + "last_sidekick": dict(last_sidekick) if last_sidekick else None, + } + + +def _session_id(hook_input: dict[str, Any]) -> str: + return str(hook_input.get("session_id") or "unknown-session") + + +def record_session_start(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + store.record_event( + repo, + session_id, + "session_start", + { + "source": hook_input.get("source", "startup"), + "model": bounded(hook_input.get("model", ""), 200), + "branch": repo.branch, + "head_sha": repo.head_sha, + }, + ) + telemetry.record( + "session_start", + repo=repo, + session_id=session_id, + trigger=bounded(str(hook_input.get("source", "startup")), 60), + model=bounded(hook_input.get("model", ""), 200), + api_key_configured=bool(api_key()), + is_git_repo=not repo.identity.startswith("local:"), + ) + + +def record_user_prompt( + store: EvidenceStore, hook_input: dict[str, Any] +) -> tuple[RepoContext, str, str, bool]: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prompt = redact(hook_input.get("prompt", "")).strip() + is_first_prompt = not store.has_event(repo.identity, session_id, "user_prompt") + store.record_event(repo, session_id, "user_prompt", {"text": prompt}) + return repo, session_id, prompt, is_first_prompt + + +def _tool_result_preview(response: Any) -> str: + if isinstance(response, dict): + selected = {} + for key in ( + "stdout", + "stderr", + "output", + "content", + "error", + "filePath", + "success", + "interrupted", + ): + if key in response: + selected[key] = response[key] + response = selected or {"keys": sorted(response.keys())[:20]} + return bounded(response, MAX_RESULT_CHARS) + + +def tool_payload(hook_input: dict[str, Any], *, failed: bool | None = False) -> dict[str, Any]: + name = str(hook_input.get("tool_name") or "unknown") + tool_input = hook_input.get("tool_input") or {} + if not isinstance(tool_input, dict): + tool_input = {} + payload: dict[str, Any] = { + "tool": name, + "failed": failed, + "duration_ms": hook_input.get("duration_ms"), + "agent_role": "sidekick" if hook_input.get("agent_id") else "main", + } + if hook_input.get("agent_id"): + payload["agent_id"] = bounded(hook_input["agent_id"], 200) + if hook_input.get("agent_type"): + payload["agent_type"] = bounded(hook_input["agent_type"], 200) + + if name in {"Read", "Write", "Edit", "MultiEdit", "NotebookEdit"}: + path = tool_input.get("file_path") or tool_input.get("notebook_path") + if path: + payload["path"] = bounded(path, 1000) + if name in {"Write", "Edit", "MultiEdit", "NotebookEdit"}: + payload["mutation_chars"] = sum( + len(str(tool_input.get(key, ""))) + for key in ("content", "new_string", "new_source", "edits") + ) + elif name == "Bash" or "command" in tool_input: + command = bounded(tool_input.get("command", ""), MAX_COMMAND_CHARS) + payload["command"] = command + payload["command_kind"] = ( + "test" + if TEST_COMMAND_RE.search(command) + else "build" + if BUILD_COMMAND_RE.search(command) + else "shell" + ) + response = ( + hook_input.get("error") if failed else hook_input.get("tool_response") + ) + payload["result_preview"] = _tool_result_preview(response) + elif name in {"Grep", "Glob", "WebSearch", "WebFetch"}: + for key in ("pattern", "path", "query", "url"): + if tool_input.get(key): + payload[key] = bounded(tool_input[key], 1000) + else: + payload["input_keys"] = sorted(tool_input.keys())[:20] + if failed: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + + if failed and "error" not in payload: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + return payload + + +def record_tool( + store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool | None = False +) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + payload = tool_payload(hook_input, failed=failed) + if payload.get("path"): + payload["repo_path"] = _repo_relative_path(repo, str(payload["path"])) + store.record_event( + repo, + session_id, + "tool_failure" if failed else "tool_result", + payload, + ) + + +def record_sidekick_start( + store: EvidenceStore, hook_input: dict[str, Any], *, inject_context: bool = True +) -> str: + """Record a native sidekick and reuse the main turn's retrieved memories.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_id = bounded(hook_input.get("agent_id", "unknown-agent"), 200) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + context = combine_context( + format_context(store.injected_memories(session_id, repo.identity)) + ) + if not inject_context: + context = "" + first_start = store.start_sidekick( + repo, session_id, agent_id, agent_type, len(context) + ) + store.record_event( + repo, + session_id, + "sidekick_start", + { + "agent_id": agent_id, + "agent_type": agent_type, + "context_chars": len(context) if first_start else 0, + "worktree_root": bounded(repo.root, 2000), + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="start", + first_start=first_start, + context_chars=len(context) if first_start else 0, + ) + return context if first_start else "" + + +def record_sidekick_stop(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + agent_id = bounded(hook_input.get("agent_id", ""), 200) + final_message = redact(hook_input.get("last_assistant_message", "")).strip() + transcript_path = bounded(hook_input.get("agent_transcript_path", ""), 2000) + agent_id = store.stop_sidekick( + repo, + session_id, + agent_id, + agent_type, + transcript_path, + final_message, + ) + store.record_event( + repo, + session_id, + "sidekick_stop", + { + "agent_id": agent_id, + "agent_type": agent_type, + "transcript_path": transcript_path, + "final_message": final_message, + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="stop", + has_transcript=bool(transcript_path), + message_chars=len(final_message), + ) + + +def _ordered_unique(values: Iterable[str]) -> list[str]: + seen: set[str] = set() + result = [] + for value in values: + if value and value not in seen: + seen.add(value) + result.append(value) + return result + + +def _repo_relative_path(repo: RepoContext, value: str) -> str: + value = str(value or "").strip() + if not value: + return "" + try: + path = Path(value) + if path.is_absolute(): + try: + return path.resolve().relative_to(Path(repo.root).resolve()).as_posix() + except ValueError: + return "" + except (OSError, ValueError): + pass + return bounded(value, 1000) + + +def _render_command_lines(commands: list[dict[str, str]]) -> list[str]: + lines = [] + for command in commands: + line = f"- [{command['status']}/{command['kind']}] {command['command']}" + if command["result"]: + line += f" — {bounded(command['result'], 500).replace(chr(10), ' ')}" + lines.append(line) + return lines + + +def build_episode( + repo: RepoContext, + session_id: str, + packet_id: str, + events: list[dict[str, Any]], + *, + canonical_task: str = "", + task_outcome: str = "", +) -> tuple[str, dict[str, Any]]: + prompts = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "user_prompt" and e["payload"].get("text") + ] + assistant_conclusions = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "assistant_stop" and e["payload"].get("text") + ] + sidekick_outcomes = [ + redact(e["payload"].get("final_message", "")).strip() + for e in events + if e["kind"] == "sidekick_stop" and e["payload"].get("final_message") + ] + tools = [ + e["payload"] for e in events if e["kind"] in {"tool_result", "tool_failure"} + and e["payload"].get("agent_role", "main") == "main" + ] + read_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") == "Read" + ) + modified_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") in {"Write", "Edit", "MultiEdit", "NotebookEdit"} + ) + searches = [ + {key: t[key] for key in ("tool", "pattern", "path", "query", "url") if key in t} + for t in tools + if t.get("tool") in {"Grep", "Glob", "WebSearch", "WebFetch"} + ] + commands = [ + { + "command": t.get("command", ""), + "kind": t.get("command_kind", "shell"), + "status": "unknown" if t.get("failed", False) is None else "failed" if t.get("failed") else "succeeded", + "result": t.get("result_preview", ""), + } + for t in tools + if t.get("command") + ] + + task = bounded(canonical_task or (prompts[0] if prompts else ""), 4000) + conclusion = redact(assistant_conclusions[-1] if assistant_conclusions else "").strip() + outcome = bounded(task_outcome, 2000) + + extraction_messages: list[dict[str, str]] = [] + pending_user_messages: list[dict[str, str]] = [] + if task and not prompts: + pending_user_messages.append({"role": "user", "content": task}) + for event in events: + if event["kind"] == "user_prompt" and event["payload"].get("text"): + pending_user_messages.append( + { + "role": "user", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + elif event["kind"] == "assistant_stop": + transcript_messages = event["payload"].get("transcript_messages") or [] + if isinstance(transcript_messages, list) and transcript_messages: + transcript_users = { + redact(message.get("content") or "").strip() + for message in transcript_messages + if isinstance(message, dict) and message.get("role") == "user" + } + extraction_messages.extend( + message + for message in pending_user_messages + if message["content"].strip() not in transcript_users + ) + extraction_messages.extend( + { + "role": str(message.get("role") or ""), + "content": redact(message.get("content") or "").strip(), + } + for message in transcript_messages + if isinstance(message, dict) + and message.get("role") in {"user", "assistant"} + and message.get("content") + ) + else: + extraction_messages.extend(pending_user_messages) + if event["payload"].get("text"): + extraction_messages.append( + { + "role": "assistant", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + pending_user_messages = [] + elif event["kind"] == "sidekick_stop": + pass + extraction_messages.extend(pending_user_messages) + + structured = { + "packet_id": packet_id, + "repo": repo.identity, + "app_id": repo.app_id, + "session_id": session_id, + "branch": repo.branch, + "head_sha": repo.head_sha, + "task": task, + "task_outcome": outcome, + "assistant_conclusion": conclusion, + "user_messages": prompts, + "assistant_outcomes": assistant_conclusions, + "sidekick_outcomes": sidekick_outcomes, + "extraction_messages": extraction_messages, + "files_read": read_paths[:50], + "files_modified": modified_paths[:50], + "searches": searches[-30:], + "commands": commands[-30:], + } + lines = ["Coding-session episode"] + if task: + lines.extend(["", "Task:", task]) + if modified_paths: + lines.extend( + ["", "Files modified:", *[f"- {path}" for path in modified_paths[:50]]] + ) + if read_paths: + lines.extend(["", "Files read:", *[f"- {path}" for path in read_paths[:50]]]) + if commands: + lines.append("") + lines.append("Observed commands:") + lines.extend(_render_command_lines(commands[-30:])) + if searches: + lines.extend( + [ + "", + "Observed searches:", + *[ + f"- {json.dumps(item, ensure_ascii=False, sort_keys=True)}" + for item in searches[-20:] + ], + ] + ) + if conclusion: + lines.extend(["", "Agent conclusion:", conclusion]) + if outcome: + lines.extend(["", "Task outcome:", outcome]) + lines.extend( + [ + "", + f"Provenance: repo={repo.identity}; branch={repo.branch}; head={repo.head_sha}; packet={packet_id}", + ] + ) + content = "\n".join(lines) + return bounded(content, MAX_EPISODE_CHARS), structured + + +def build_semantic_evidence(structured: dict[str, Any]) -> str: + """Format changed paths for memory extraction. + + Test and build results remain in the local evidence store for diagnostics, + but are not useful repository knowledge by default and should not steer + memory extraction toward transient verification details. + """ + modified_paths = [ + bounded(path, 500) for path in structured.get("files_modified", [])[:20] + ] + commands = structured.get("commands") or [] + if not any(command.get("status") == "failed" for command in commands): + commands = [] + + if not modified_paths and not commands: + return "" + + lines = ["Additional repository details from this session"] + if modified_paths: + lines.extend( + [ + "", + "Changed paths:", + *[f"- {path}" for path in modified_paths], + ] + ) + if commands: + lines.extend(["", "Commands run in this session:", *_render_command_lines(commands)]) + return bounded("\n".join(lines), 8000) + + +def build_extraction_messages(structured: dict[str, Any]) -> list[dict[str, str]]: + """Build the session messages sent to Mem0 for memory extraction.""" + evidence = build_semantic_evidence(structured) + messages = [ + {"role": message["role"], "content": redact(message["content"]).strip()} + for message in structured.get("extraction_messages", []) + if message.get("role") in {"user", "assistant"} and message.get("content") + ] + if evidence: + for message in reversed(messages): + if message["role"] == "assistant": + message["content"] = f"{message['content']}\n\n{evidence}" + break + else: + messages.append({"role": "assistant", "content": evidence}) + + return messages + + +def _estimated_tokens(value: str) -> int: + """Conservatively estimate tokens without adding a tokenizer dependency.""" + ascii_chars = sum(ord(char) < 128 for char in value) + return math.ceil((ascii_chars * 0.4) + (len(value) - ascii_chars)) + + +def _message_tokens(messages: list[dict[str, str]]) -> int: + return _estimated_tokens(json.dumps(messages, ensure_ascii=False)) + + +def _is_agent_assignment(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent assignment (") + + +def _is_agent_response(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent response (") + + +def extraction_message_batches( + messages: list[dict[str, str]], + *, + max_tokens: int = MAX_EXTRACTION_INPUT_TOKENS, +) -> list[list[dict[str, str]]]: + """Keep exchanges together when possible; split oversized messages to enforce the request budget.""" + if not messages or _message_tokens(messages) <= max_tokens: + return [messages] + + exchanges: list[list[dict[str, str]]] = [] + exchange: list[dict[str, str]] = [] + for message in messages: + if message.get("role") == "user" and exchange: + exchanges.append(exchange) + exchange = [] + exchange.append(message) + if exchange: + exchanges.append(exchange) + + units: list[list[dict[str, str]]] = [] + for exchange in exchanges: + if _message_tokens(exchange) <= max_tokens: + units.append(exchange) + continue + index = 0 + while index < len(exchange): + message = exchange[index] + if ( + _is_agent_assignment(message) + and index + 1 < len(exchange) + and _is_agent_response(exchange[index + 1]) + ): + units.append(exchange[index : index + 2]) + index += 2 + else: + units.append([message]) + index += 1 + + bounded_units: list[list[dict[str, str]]] = [] + for unit in units: + if _message_tokens(unit) <= max_tokens: + bounded_units.append(unit) + continue + for message in unit: + remaining = message["content"] + while remaining: + low, high = 0, len(remaining) + while low < high: + middle = (low + high + 1) // 2 + if _message_tokens([{**message, "content": remaining[:middle]}]) <= max_tokens: + low = middle + else: + high = middle - 1 + if low == 0: + raise ValueError("Extraction token budget cannot fit a message") + bounded_units.append([{**message, "content": remaining[:low]}]) + remaining = remaining[low:] + + batches: list[list[dict[str, str]]] = [] + batch: list[dict[str, str]] = [] + for unit in bounded_units: + candidate = [*batch, *unit] + if batch and _message_tokens(candidate) > max_tokens: + batches.append(batch) + batch = list(unit) + else: + batch = candidate + if batch: + batches.append(batch) + return batches + + +def _request_json( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + raw = json.dumps(payload, ensure_ascii=False).encode() + request = urllib.request.Request( + url, + data=raw, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + parsed = json.loads(response_raw or b"{}") + return parsed, len(raw), len(response_raw) + + +def _request_json_with_network_retry( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + """Retry one transient connection failure without retrying API responses.""" + try: + return _request_json(url, key, payload, timeout) + except urllib.error.HTTPError: + raise + except (urllib.error.URLError, TimeoutError, OSError): + time.sleep(0.25) + return _request_json(url, key, payload, timeout) + + +def _get_json( + url: str, key: str, timeout: float +) -> tuple[dict[str, Any] | list[Any], int]: + request = urllib.request.Request( + url, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="GET", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + return json.loads(response_raw or b"{}"), len(response_raw) + + +def _event_id(response: dict[str, Any] | list[Any]) -> str: + return str(response.get("event_id", "")) if isinstance(response, dict) else "" + + +def _stored_event_ids(value: Any) -> list[str]: + raw = str(value or "") + if not raw.startswith("["): + return [] + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return [] + return [str(item or "") for item in parsed] if isinstance(parsed, list) else [] + + +def _result_count(response: dict[str, Any] | list[Any]) -> int: + if isinstance(response, dict): + results = response.get("results") + return len(results) if isinstance(results, list) else 0 + return len(response) if isinstance(response, list) else 0 + + +def touch_handoff_heartbeat() -> None: + """Mark the worker's handoff file alive so recovery does not relaunch it.""" + path = os.environ.get("MEM0_CODE_HANDOFF_PATH", "") + if not path: + return + try: + os.utime(path) + except OSError: + pass + + +def _wait_for_event(api_url: str, key: str, event_id: str) -> tuple[str, int, int]: + """Wait for extraction to finish before a later task can search the store.""" + if not event_id: + return "MISSING", 0, 0 + wait_seconds = float(os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120")) + poll_seconds = max(float(os.environ.get("MEM0_CODE_EVENT_POLL_SECONDS", "1")), 0.1) + deadline = time.monotonic() + wait_seconds + response_chars = 0 + while time.monotonic() < deadline: + touch_handoff_heartbeat() + try: + response, size = _get_json( + f"{api_url}/v1/event/{event_id}/", + key, + min(10, poll_seconds + 5), + ) + except urllib.error.HTTPError as exc: + if exc.code not in {408, 429} and exc.code < 500: + raise + time.sleep(poll_seconds) + continue + except (urllib.error.URLError, TimeoutError, OSError): + # The extraction job is durable server-side. A transient polling + # failure must not discard a job that may still complete normally. + time.sleep(poll_seconds) + continue + response_chars += size + status = ( + str(response.get("status", "UNKNOWN")) + if isinstance(response, dict) + else "UNKNOWN" + ) + if status in {"SUCCEEDED", "FAILED"}: + return status, response_chars, _result_count(response) + time.sleep(poll_seconds) + return "TIMEOUT", response_chars, 0 + + +def _record_flush( + repo: RepoContext, + session_id: str, + reason: str, + status: str, + elapsed: float, + **extra: Any, +) -> None: + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status=status, + success=status in {"semantic-succeeded", "nothing-to-flush"}, + duration_ms=round(elapsed, 2), + **extra, + ) + + +def flush_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + key = api_key() + if not key: + telemetry.record("flush", reason=reason, status="local-only", success=False) + return {"status": "local-only", "reason": "no-api-key"} + + session_id = _session_id(hook_input) + if session_id == "unknown-session": + telemetry.record("flush", reason=reason, status="no-session-id", success=False) + return {"status": "error", "reason": "no-session-id"} + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prepared = store.prepare_flush(repo, session_id, reason) + if prepared is None: + return {"status": "nothing-to-flush"} + packet_id, events = prepared + existing_flush = store.flush_record(packet_id) or {} + + _, structured = build_episode( + repo, + session_id, + packet_id, + events, + canonical_task=bounded(hook_input.get("task", ""), 4000), + task_outcome=bounded(hook_input.get("task_outcome", ""), 2000), + ) + + metadata = {"source": _harness_source_tag} + if repo.branch and repo.branch not in {"detached", "unknown"}: + metadata["branch"] = repo.branch + if repo.head_sha: + metadata["git_sha"] = repo.head_sha + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + add_url = f"{api_url}/v3/memories/add/" + + write_user = _scope_value(user_id()) + write_project = _scope_value(repo.project_id) + if not write_user or not _scope_value(repo.app_id) or not write_project: + telemetry.record("flush", reason=reason, status="unscoped", success=False) + return {"status": "error", "reason": "wildcard-scope"} + + body = { + "agent_id": write_project, + "user_id": write_user, + "app_id": repo.app_id, + "run_id": session_id, + "metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)}, + "agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS, + "custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS, + "custom_categories": CODING_MEMORY_CATEGORIES, + "infer": True, + } + + started = time.perf_counter() + try: + stored_events = _stored_event_ids(existing_flush.get("semantic_event_id")) + existing_event = ( + "" if stored_events else str(existing_flush.get("semantic_event_id") or "") + ) + if existing_event: + existing_status, existing_resp, existing_items = _wait_for_event( + api_url, key, existing_event + ) + if existing_status == "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=existing_event, + error="", + ) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + True, + item_count=existing_items, + response_chars=existing_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=existing_items, + resumed=True, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "semantic_status": existing_status, + "memory_count": existing_items, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + if existing_status == "TIMEOUT": + elapsed = (time.perf_counter() - started) * 1000 + error = "semantic extraction event timed out" + store.update_flush(packet_id, status="semantic-timeout", error=error) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + False, + response_chars=existing_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-timeout", + elapsed, + resumed=True, + error_kind="timeout", + ) + return { + "status": "semantic-timeout", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + + message_batches = [ + batch + for batch in extraction_message_batches( + build_extraction_messages(structured) + ) + if batch + ] + batches = [(body, messages) for messages in message_batches] + if not batches: + store.update_flush(packet_id, status="semantic-succeeded", error="") + return {"status": "nothing-to-flush", "packet_id": packet_id} + operation_name = "flush-retry" if stored_events else "flush" + semantic_events = stored_events[: len(batches)] + semantic_events += [""] * (len(batches) - len(semantic_events)) + semantic_req = 0 + semantic_resp = 0 + for index, (body, messages) in enumerate(batches): + if semantic_events[index]: + continue + semantic_response, request_chars, response_chars = _request_json( + add_url, + key, + {**body, "messages": messages}, + 15, + ) + semantic_events[index] = _event_id(semantic_response) + semantic_req += request_chars + semantic_resp += response_chars + store.update_flush( + packet_id, + status="semantic-queued", + semantic_event_id=json.dumps(semantic_events), + ) + + semantic_event = semantic_events[-1] + semantic_status = "SUCCEEDED" + event_resp = 0 + semantic_items = 0 + failed_event = semantic_event + for index, queued_event in enumerate(semantic_events): + status, response_chars, item_count = _wait_for_event( + api_url, key, queued_event + ) + event_resp += response_chars + semantic_items += item_count + if status != "SUCCEEDED": + semantic_status = status + failed_event = queued_event + if status in {"FAILED", "MISSING"}: + semantic_events[index] = "" + store.update_flush( + packet_id, semantic_event_id=json.dumps(semantic_events) + ) + break + if semantic_status != "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + error = f"semantic extraction event {semantic_status.lower()}" + store.update_flush( + packet_id, status=f"semantic-{semantic_status.lower()}", error=error + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + False, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + f"semantic-{semantic_status.lower()}", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + error_kind=telemetry.error_kind(error), + ) + return { + "status": f"semantic-{semantic_status.lower()}", + "packet_id": packet_id, + "semantic_event_id": failed_event, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=json.dumps(semantic_events), + error="", + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + True, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + request_chars=semantic_req, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": semantic_event, + "semantic_status": semantic_status, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + except Exception as exc: # hooks must fail open + elapsed = (time.perf_counter() - started) * 1000 + error = bounded(str(exc), 1000) + store.update_flush(packet_id, status="error", error=error) + store.operation(repo, session_id, "flush", elapsed, False, error=error) + _record_flush( + repo, + session_id, + reason, + "error", + elapsed, + error_kind=telemetry.error_kind(exc), + ) + return {"status": "error", "packet_id": packet_id, "error": error} + + +def checkpoint_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + """Run remote extraction at a durable boundary.""" + return flush_session(store, hook_input, reason) + + + +def search_memories( + store: EvidenceStore | None, + repo: RepoContext, + session_id: str | None, + query: str, + *, + top_k: int | None = None, + category: str | None = None, + scope: str | None = None, + run_id: str | None = None, + operation: str = "search", + timeout: float = 5, +) -> MemorySearchResult: + key = api_key() + if not key or not query.strip(): + return MemorySearchResult(False, 0, 0, []) + search_once = os.environ.get( + "MEM0_CODE_SEARCH_ONCE_PER_SESSION", "false" + ).lower() in { + "1", + "true", + "yes", + "on", + } + track_session = store is not None and bool(session_id) + if ( + search_once + and track_session + and store.has_operation(repo.identity, session_id, "search") + ): + return MemorySearchResult(False, 0, 0, []) + + result_limit = min( + max( + top_k + if top_k is not None + else _int_option("top_k", "MEM0_CODE_TOP_K", 3), + 1, + ), + 20, + ) + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ValueError(f"Unknown memory category: {category}") + user, project = _scope_value(user_id()), _scope_value(repo.project_id) + if not user or not project or not _scope_value(repo.app_id): + return MemorySearchResult(False, 0, 0, []) + filters = _search_filters(user, repo, resolve_search_scope(scope)) + if category: + filters = {"AND": [filters, {"categories": {"contains": category}}]} + if run_id: + filters = {"AND": [filters, {"run_id": run_id}]} + payload = { + "query": query, + "app_id": repo.app_id, + "filters": filters, + "top_k": result_limit, + "rerank": False, + "latest_only": True, + } + url = ( + os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + + "/v3/memories/search/" + ) + started = time.perf_counter() + try: + response, request_chars, response_chars = _request_json_with_network_retry( + url, key, payload, timeout + ) + memories = ( + response if isinstance(response, list) else response.get("results", []) + ) + memories = [ + memory + for memory in memories + if isinstance(memory, dict) + and (memory.get("metadata") or {}).get("record_kind") != "task_episode" + ][:result_limit] + if track_session: + returned_memories = store.unseen(session_id, repo.identity, memories) + store.mark_injected(session_id, repo.identity, returned_memories) + already_shown_count = len(memories) - len(returned_memories) + else: + returned_memories = memories + already_shown_count = 0 + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation( + repo, + session_id, + operation, + elapsed, + True, + item_count=len(returned_memories), + request_chars=request_chars, + response_chars=response_chars, + ) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=True, + duration_ms=round(elapsed, 2), + matched_count=len(memories), + returned_count=len(returned_memories), + already_shown_count=already_shown_count, + top_k=result_limit, + has_category=bool(category), + ) + return MemorySearchResult( + succeeded=True, + matched_count=len(memories), + already_shown_count=already_shown_count, + memories=returned_memories, + ) + except Exception as exc: + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation(repo, session_id, operation, elapsed, False, error=str(exc)) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=False, + duration_ms=round(elapsed, 2), + top_k=result_limit, + has_category=bool(category), + error_kind=telemetry.error_kind(exc), + ) + return MemorySearchResult(False, 0, 0, []) + + +def format_context( + memories: list[dict[str, Any]], + heading: str = "Relevant repository memories:", +) -> str: + if not memories: + return "" + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + lines = [heading] if heading else [] + for memory in memories: + text = re.sub( + r"\s+", + " ", + redact(memory.get("memory") or memory.get("text") or ""), + ) + text = text.strip() + if not text: + continue + branch = str((memory.get("metadata") or {}).get("branch") or "").strip() + branch_label = ( + f" [learnt on branch {branch}]" + if branch.casefold() not in {"", "main", "master", "unknown", "detached"} + else "" + ) + number = len(lines) if heading else len(lines) + 1 + entry = f"{number}. {text}{branch_label}" + candidate = "\n".join([*lines, entry]) + if len(candidate) <= limit: + lines.append(entry) + continue + if not lines or (heading and len(lines) == 1): + prefix = f"{number}. " + suffix = f"…{branch_label}" + available = ( + limit + - len("\n".join(lines)) + - (1 if lines else 0) + - len(prefix) + - len(suffix) + ) + if available > 0: + lines.append(prefix + text[:available].rstrip() + suffix) + break + minimum_lines = 2 if heading else 1 + return "\n".join(lines) if len(lines) >= minimum_lines else "" + + +def format_search_result(result: MemorySearchResult) -> str: + """Return only the text the coding agent needs from an explicit memory search.""" + if not result.succeeded: + return "Memory search failed." + if result.memories: + rendered = format_context(result.memories, heading="") + if rendered: + return rendered + return "No matching memories found." + + +def combine_context(*contexts: str) -> str: + """Combine memory sources under one hard budget without repeated lines.""" + seen: set[str] = set() + lines: list[str] = [] + for context in contexts: + for line in str(context or "").splitlines(): + normalized = re.sub(r"\s+", " ", line).strip().casefold() + if not normalized or normalized in seen: + continue + seen.add(normalized) + lines.append(line.rstrip()) + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + return bounded("\n".join(lines), limit) if lines else "" + + +def _scoped_memory_ids( + api_url: str, key: str, user: str, repo: RepoContext, include_project: bool +) -> list[str]: + """List this user's memory ids for this repository, plus the shared project memory when asked.""" + ids: list[str] = [] + seen: set[str] = set() + prefix = repo.app_id + _collect_memory_ids( + api_url, key, {"user_id": user}, ids, seen, + app_id_prefix=prefix, + ) + if include_project: + for project_id in _shared_project_ids(repo): + _collect_memory_ids( + api_url, key, {"agent_id": project_id}, ids, seen, + app_id_prefix=prefix, + ) + return ids + + +def _collect_memory_ids( + api_url: str, + key: str, + filters: dict[str, Any], + ids: list[str], + seen: set[str], + *, + app_id_prefix: str = "", +) -> None: + """Page through one list filter; the list endpoint returns nothing for an OR whose user branch has no memories.""" + payload = {"filters": filters} + for page in range(1, FORGET_MAX_PAGES + 1): + parsed, _, _ = _request_json( + f"{api_url}/v2/memories/?page={page}&page_size={FORGET_PAGE_SIZE}", + key, + payload, + 15, + ) + items = parsed.get("results") if isinstance(parsed, dict) else parsed + if not isinstance(items, list) or not items: + break + for item in items: + if not isinstance(item, dict): + continue + memory_id = str(item.get("id", "")) + if not memory_id or memory_id in seen: + continue + if app_id_prefix: + item_app_id = str(item.get("app_id") or "") + if item_app_id != app_id_prefix and not item_app_id.startswith(app_id_prefix + "/"): + continue + seen.add(memory_id) + ids.append(memory_id) + if len(items) < FORGET_PAGE_SIZE: + break + + +def _delete_memory(api_url: str, key: str, memory_id: str) -> bool: + request = urllib.request.Request( + f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/", + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="DELETE", + ) + try: + with urllib.request.urlopen(request, timeout=15): + return True + except Exception: + return False + + +def forget_remote_repo( + repo: RepoContext, *, include_project_memory: bool = False +) -> dict[str, Any]: + """Delete this user's memories for this repository; project memory is shared, so only on request.""" + key = api_key() + if not key: + telemetry.record("forget", repo=repo, success=False, error_kind="no-api-key") + return {"status": "error", "error": "Mem0 API key is not configured"} + user = _scope_value(user_id()) + if not user or not _scope_value(repo.app_id) or not _scope_value(repo.project_id): + telemetry.record("forget", repo=repo, success=False, error_kind="unscoped") + return { + "status": "error", + "error": "Refusing to forget: the user or repository scope is a wildcard", + } + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + try: + memory_ids = _scoped_memory_ids(api_url, key, user, repo, include_project_memory) + except Exception as exc: + telemetry.record( + "forget", repo=repo, success=False, error_kind=telemetry.error_kind(exc) + ) + return {"status": "error", "error": bounded(str(exc), 1000)} + deleted = sum(_delete_memory(api_url, key, memory_id) for memory_id in memory_ids) + failed = len(memory_ids) - deleted + telemetry.record("forget", repo=repo, success=not failed, item_count=deleted) + if failed: + return { + "status": "partial", + "deleted": deleted, + "failed": failed, + "error": f"{failed} of {len(memory_ids)} memories could not be deleted", + } + return {"status": "deleted", "deleted": deleted} + + +def _doctor_mem0_authentication(repo: RepoContext) -> dict[str, Any]: + """Verify the configured key with one read-only, repository-scoped search.""" + key = api_key() + if not key: + return {"ok": False, "detail": "API key missing"} + payload = { + "query": "Mem0 authentication check", + "filters": { + "AND": [ + {"user_id": user_id()}, + {"app_id": repo.app_id}, + ] + }, + "top_k": 1, + "threshold": 1.0, + "rerank": False, + } + url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + started = time.perf_counter() + try: + _request_json(f"{url}/v3/memories/search/", key, payload, 5) + except Exception as exc: + return {"ok": False, "detail": bounded(str(exc), 300)} + elapsed = (time.perf_counter() - started) * 1000 + return {"ok": True, "detail": f"connected ({elapsed:.0f} ms)"} + + +def _doctor_user_id() -> dict[str, Any]: + """Flag a configured user ID the plugin refuses, since the silent fallback surprises people.""" + configured = _plugin_option("user_id", "MEM0_CODE_USER_ID") or os.environ.get( + "MEM0_USER_ID", "" + ) + if configured and not _scope_value(configured): + return { + "ok": False, + "detail": f"configured user_id {configured!r} is a wildcard; using {user_id()!r}", + } + return {"ok": True, "detail": user_id()} + + +def doctor(cwd: str | None = None) -> dict[str, Any]: + repo = resolve_repo(cwd) + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + checks: dict[str, dict[str, Any]] = { + "python": { + "ok": tuple(sys.version_info[:2]) >= (3, 10), + "detail": f"{sys.version_info.major}.{sys.version_info.minor}", + }, + "data_directory": { + "ok": os.access(directory, os.W_OK), + "detail": str(directory), + }, + "mem0_api_key": { + "ok": bool(api_key()), + "detail": "configured" if api_key() else "missing", + }, + "repository": { + "ok": bool(repo.identity), + "detail": repo.identity, + }, + "user_id": _doctor_user_id(), + "mem0_authentication": _doctor_mem0_authentication(repo), + } + return { + "ok": all(bool(value["ok"]) for value in checks.values()), + "plugin_version": PLUGIN_VERSION, + "repo_id": repo.identity, + "app_id": repo.app_id, + "user_id": user_id(), + "checks": checks, + } diff --git a/integrations/agent-plugin-core/python/telemetry.py b/integrations/agent-plugin-core/python/telemetry.py new file mode 100644 index 000000000..249595475 --- /dev/null +++ b/integrations/agent-plugin-core/python/telemetry.py @@ -0,0 +1,397 @@ +#!/usr/bin/env python3 +"""Anonymous usage telemetry for Mem0 agent plugins. + +Hooks run on a 3-6 second budget and fire on every tool call, so recording never +touches the network: `record` appends one JSON line to a local spool and returns. +A detached `python3 telemetry.py` drains the spool in one batched PostHog request, +started once per session and again from the flush worker that is already detached. + +Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false. + +Never sends prompts, memory text, queries, file paths, repository names, or API +keys: only event names, durations, counts, coarse outcomes, and salted hashes. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import platform +import subprocess +import sys +import time +import urllib.error +import urllib.request +import uuid +from pathlib import Path +from typing import Any + +import memory_core + +_harness: str = "generic" +_source_tag: str = "MEM0_PLUGIN" +_PRIVATE_KEYS = { + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +} + + +def init(harness: str = "generic", source_tag: str = "") -> None: + global _harness, _source_tag + _harness = harness + _source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN" + +POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" +POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" +POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/" +EVENT_PREFIX = "code" +SPOOL_LIMIT_BYTES = 256 * 1024 +BATCH_SIZE = 100 +SEND_TIMEOUT = 5 +CLAIM_STALE_SECONDS = 120 +CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60 + + +def is_enabled() -> bool: + """Whether telemetry is switched on for this process.""" + return os.environ.get("MEM0_TELEMETRY", "true").strip().lower() not in { + "false", + "0", + "no", + "off", + } + + +def _digest(value: str, length: int = 16) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length] + + +def _safe_value(value: Any) -> Any: + if isinstance(value, str): + return memory_core.redact(value) + if isinstance(value, dict): + return { + key: _safe_value(item) + for key, item in value.items() + if "".join(character for character in str(key).lower() if character.isalnum()) + not in _PRIVATE_KEYS + } + if isinstance(value, (list, tuple)): + return [_safe_value(item) for item in value] + if value is None or isinstance(value, (bool, int, float)): + return value + return memory_core.redact(value) + + +def _spool_path() -> Path: + return memory_core.data_dir() / "telemetry.jsonl" + + +def _identity_path() -> Path: + return memory_core.data_dir() / "telemetry-identity.json" + + +def _read_identity() -> dict[str, str]: + try: + value = json.loads(_identity_path().read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + return value if isinstance(value, dict) else {} + + +def _write_identity(identity: dict[str, str]) -> None: + path = _identity_path() + temporary = path.with_suffix(f".{os.getpid()}.tmp") + try: + path.parent.mkdir(parents=True, exist_ok=True) + temporary.write_text(json.dumps(identity), encoding="utf-8") + temporary.replace(path) + except OSError: + try: + temporary.unlink() + except OSError: + pass + + +def anonymous_id(identity: dict[str, str] | None = None) -> str: + """Per-machine anonymous identifier, created and persisted on first use.""" + identity = _read_identity() if identity is None else identity + existing = identity.get("anonymous_id") + if existing: + return existing + created = f"code-anon-{uuid.uuid4().hex}" + identity["anonymous_id"] = created + _write_identity(identity) + return created + + +def is_first_run() -> bool: + """Whether this machine has never recorded a plugin event before.""" + return not _identity_path().exists() + + +def record( + event: str, + *, + repo: Any = None, + session_id: str | None = None, + **properties: Any, +) -> None: + """Append one event to the local spool. Never blocks and never raises.""" + if not is_enabled(): + return + try: + spool = _spool_path() + try: + if spool.stat().st_size > SPOOL_LIMIT_BYTES: + return + except OSError: + pass + properties = _safe_value(properties) + properties.update( + harness=_harness, + plugin_version=memory_core.PLUGIN_VERSION, + os=sys.platform, + python_version=platform.python_version(), + ) + if repo is not None: + properties["repo_hash"] = _digest(getattr(repo, "identity", "")) + if session_id: + properties["session_hash"] = _digest(session_id) + line = json.dumps( + { + "event": f"{EVENT_PREFIX}.{event}", + "timestamp": memory_core.utc_now(), + "properties": { + key: value for key, value in properties.items() if value is not None + }, + }, + separators=(",", ":"), + default=str, + ) + spool.parent.mkdir(parents=True, exist_ok=True) + with spool.open("a", encoding="utf-8") as handle: + handle.write(line + "\n") + except Exception: + pass + + +def error_kind(exc: BaseException | str) -> str: + """Coarse, content-free label for a failure, safe to send.""" + text = exc if isinstance(exc, str) else f"{type(exc).__name__}: {exc}" + lowered = text.lower() + if "timed out" in lowered or "timeout" in lowered: + return "timeout" + if "401" in lowered or "403" in lowered or "unauthor" in lowered or "forbidden" in lowered: + return "auth" + if "429" in lowered or "rate limit" in lowered: + return "rate-limited" + if any(code in lowered for code in ("500", "502", "503", "504")): + return "server-error" + if "400" in lowered or "422" in lowered: + return "bad-request" + if isinstance(exc, str): + return "other" + if isinstance(exc, urllib.error.URLError): + return "network" + return type(exc).__name__ + + +def spawn_flush() -> bool: + """Start the detached sender that drains the spool.""" + if not is_enabled(): + return False + try: + if not _spool_path().exists() and not any( + memory_core.data_dir().glob("telemetry-*.sending") + ): + return False + subprocess.Popen( + [sys.executable, str(Path(__file__).resolve())], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + close_fds=True, + **memory_core.detached_process_kwargs(), + ) + return True + except Exception: + return False + + +def _claim_spool() -> Path | None: + """Rename the spool aside so exactly one sender owns each batch.""" + directory = memory_core.data_dir() + claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending" + spool = _spool_path() + try: + spool.replace(claim) + return claim + except OSError: + pass + now = time.time() + for orphan in sorted(directory.glob("telemetry-*.sending")): + try: + age = now - orphan.stat().st_mtime + except OSError: + continue + if age > CLAIM_EXPIRY_SECONDS: + try: + orphan.unlink() + except OSError: + pass + continue + if age < CLAIM_STALE_SECONDS: + continue + try: + orphan.replace(claim) + return claim + except OSError: + continue + return None + + +def _resolve_email(key: str) -> str: + """Trade the API key for the account email so events join other Mem0 surfaces.""" + url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/" + request = urllib.request.Request( + url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json"} + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT) as response: + payload = json.loads(response.read().decode("utf-8")) + except Exception: + return "" + email = payload.get("user_email") if isinstance(payload, dict) else "" + return email if isinstance(email, str) else "" + + +def _post(payload: dict[str, Any], url: str) -> bool: + request = urllib.request.Request( + url, + data=json.dumps(payload, default=str).encode("utf-8"), + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT): + return True + except Exception: + return False + + +def resolve_distinct_id() -> tuple[str, str]: + """Return the PostHog distinct id and the anonymous id it replaced, if any.""" + identity = _read_identity() + email = identity.get("email", "") + if email: + return email, "" + key = memory_core.api_key() + if not key: + return anonymous_id(identity), "" + email = _resolve_email(key) + if not email: + return anonymous_id(identity), "" + previous = identity.get("anonymous_id", "") + identity["email"] = email + _write_identity(identity) + return email, previous + + +def flush() -> int: + """Drain claimed spools to PostHog and return the number of events sent.""" + if not is_enabled(): + return 0 + claim = _claim_spool() + if claim is None: + return 0 + try: + lines = claim.read_text(encoding="utf-8").splitlines() + except OSError: + return 0 + events = [] + for line in lines: + try: + value = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(value, dict) and value.get("event"): + events.append(value) + if not events: + try: + claim.unlink() + except OSError: + pass + return 0 + + distinct_id, aliased_anonymous_id = resolve_distinct_id() + if aliased_anonymous_id: + _post( + { + "api_key": POSTHOG_API_KEY, + "event": "$identify", + "distinct_id": distinct_id, + "properties": { + "$anon_distinct_id": aliased_anonymous_id, + "$lib": "posthog-python", + }, + }, + POSTHOG_CAPTURE_URL, + ) + + sent = 0 + for start in range(0, len(events), BATCH_SIZE): + batch = [ + { + "event": event["event"], + "distinct_id": distinct_id, + "timestamp": event.get("timestamp"), + "properties": { + "source": _source_tag, + "language": "python", + "$process_person_profile": False, + "$lib": "posthog-python", + **(event.get("properties") or {}), + }, + } + for event in events[start : start + BATCH_SIZE] + ] + if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL): + return sent + sent += len(batch) + try: + claim.unlink() + except OSError: + pass + return sent + + +def main() -> int: + flush() + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception: + raise SystemExit(0) diff --git a/integrations/agent-plugin-core/requirements-dev.txt b/integrations/agent-plugin-core/requirements-dev.txt new file mode 100644 index 000000000..1fbe37b8d --- /dev/null +++ b/integrations/agent-plugin-core/requirements-dev.txt @@ -0,0 +1,3 @@ +jsonschema>=4.23,<5 +pytest>=8,<10 +skills-ref==0.1.1 diff --git a/integrations/agent-plugin-core/skills/forget/SKILL.md.tmpl b/integrations/agent-plugin-core/skills/forget/SKILL.md.tmpl new file mode 100644 index 000000000..1b16a6c79 --- /dev/null +++ b/integrations/agent-plugin-core/skills/forget/SKILL.md.tmpl @@ -0,0 +1,26 @@ +--- +name: forget +description: Delete the Mem0 memories stored for this repository and this user. Use when the user asks to forget, clear, wipe, or delete memories. +disable-model-invocation: true +--- + +# Forget this repository's memories + +This permanently deletes remote memories. Before running anything, tell the +user exactly what will be deleted: their own memories for this repository +only. The repository's project memory is shared by everyone who works in it, +so it stays unless the user explicitly asks to delete that too. + +After the user confirms, run: + +```bash +python3 "{{PLUGIN_ROOT}}/core/memory_cli.py" --harness "{{HARNESS_ID}}" {{PLUGIN_DATA_ARG}} forget --remote --yes +``` + +If the user also asked to delete the repository's shared project memory, add +`--include-project-memory` and say that this removes it for every teammate. + +Report what the command output says was deleted. If the user only wants local +data cleared (evidence log, pending queue), run the same command without +`--remote`. Never pass `--yes` before the user has confirmed in this +conversation. diff --git a/integrations/agent-plugin-core/skills/pause/SKILL.md.tmpl b/integrations/agent-plugin-core/skills/pause/SKILL.md.tmpl new file mode 100644 index 000000000..9cc720c7a --- /dev/null +++ b/integrations/agent-plugin-core/skills/pause/SKILL.md.tmpl @@ -0,0 +1,20 @@ +--- +name: pause +description: Pause Mem0 memory capture on this machine. Use when the user wants to stop memories being recorded, for example for private work or experiments. +disable-model-invocation: true +--- + +# Pause memory capture + +To pause (hooks stop capturing and sending session content; a minimal +anonymous telemetry ping still fires at session start unless +`MEM0_TELEMETRY=false`): + +```bash +python3 "{{PLUGIN_ROOT}}/core/memory_cli.py" --harness "{{HARNESS_ID}}" {{PLUGIN_DATA_ARG}} pause +``` + +Confirm the new state back to the user, and remind them that already-created +memories still exist and remain searchable. Pending unsent packets are held +while paused, not expired, and are delivered after resuming. To turn capture +back on, use `/mem0:resume`. diff --git a/integrations/agent-plugin-core/skills/remember/SKILL.md.tmpl b/integrations/agent-plugin-core/skills/remember/SKILL.md.tmpl new file mode 100644 index 000000000..7be5fc9be --- /dev/null +++ b/integrations/agent-plugin-core/skills/remember/SKILL.md.tmpl @@ -0,0 +1,21 @@ +--- +name: remember +description: Acknowledge a "remember this" request and make sure it is captured well. Use when the user explicitly asks to remember, note, or save something for future sessions. +disable-model-invocation: true +--- + +# Remember something for future sessions + +Mem0 creates memories from the session automatically — there is no separate +write command. When the user asks to remember something: + +1. Restate the fact clearly and completely in your reply, in one or two + sentences, including any names, values, or paths it depends on. Your visible + reply is what memory extraction reads, so a precise restatement is what gets + remembered. +2. Tell the user it will be saved with this session's memories when the session + ends or compacts, and that it will surface in future sessions in this + repository (they can check later with /mem0:search). + +Do not invent a storage confirmation or a memory ID — creation happens in the +background after the session. diff --git a/integrations/agent-plugin-core/skills/resume/SKILL.md.tmpl b/integrations/agent-plugin-core/skills/resume/SKILL.md.tmpl new file mode 100644 index 000000000..b8a4d5f01 --- /dev/null +++ b/integrations/agent-plugin-core/skills/resume/SKILL.md.tmpl @@ -0,0 +1,19 @@ +--- +name: resume +description: Resume Mem0 memory capture after it was paused with /mem0:pause. +disable-model-invocation: true +--- + +# Resume memory capture + +Resume memory capture for this machine. + +Run: + +```bash +python3 "{{PLUGIN_ROOT}}/core/memory_cli.py" --harness "{{HARNESS_ID}}" {{PLUGIN_DATA_ARG}} resume +``` + +Confirm to the user that capture is active again. New sessions record evidence and +create memories as normal; nothing that happened while paused is retroactively +captured. diff --git a/integrations/agent-plugin-core/skills/search/SKILL.md.tmpl b/integrations/agent-plugin-core/skills/search/SKILL.md.tmpl new file mode 100644 index 000000000..3a255b1ad --- /dev/null +++ b/integrations/agent-plugin-core/skills/search/SKILL.md.tmpl @@ -0,0 +1,28 @@ +--- +name: search +description: Search memories from earlier {{HARNESS_NAME}} sessions in this repository. Use it when earlier work may already explain the code, error, decision, or command you need, so you can avoid repeating file reads, searches, or experiments. +argument-hint: "[question] [--top-k number] [--category category-name] [--scope repo|dir|mine] [--run-id session-id]" +disable-model-invocation: true +--- + +# Search memories + +Call `search_memories` with the user's question. Treat `--top-k`, `--category`, +`--scope`, and `--run-id` as tool arguments instead of including them in the +query. + +Omit `top_k` to use Mem0's configured default. Omit `category` to search every +category; a category is a best-effort label Mem0 assigned when it saved the +memory, so if a category search misses, repeat it without the category. Omit +`scope` to use the configured default, normally `repo`: this repository's +shared memory, which everyone who works in it contributes to, plus your own +preferences. + +Pass `scope` when the question needs something else: `dir` to narrow the +shared memory to the directory you are working in (a package inside a +monorepo), `mine` for your own preferences alone. + +Pass `run_id` with any scope to retrieve memories saved in a specific coding-agent +session. Omit `run_id` to search across sessions. It filters the memories returned; +it does not identify the session making the search request. Use a known session ID, +never invent one. Return the tool's result directly. diff --git a/integrations/agent-plugin-core/skills/status/SKILL.md.tmpl b/integrations/agent-plugin-core/skills/status/SKILL.md.tmpl new file mode 100644 index 000000000..c1ac9e8f9 --- /dev/null +++ b/integrations/agent-plugin-core/skills/status/SKILL.md.tmpl @@ -0,0 +1,23 @@ +--- +name: status +description: Show whether Mem0 memory is working in this repository, covering configuration, capture state, pending flushes, and whether the Mem0 API key is valid. Use when the user asks whether memory is on, why a memory is missing, or anything looks broken. +disable-model-invocation: false +--- + +# Memory status + +Run both commands and report the combined result in plain language: + +```bash +python3 "{{PLUGIN_ROOT}}/core/memory_cli.py" --harness "{{HARNESS_ID}}" {{PLUGIN_DATA_ARG}} status --json +python3 "{{PLUGIN_ROOT}}/core/memory_cli.py" --harness "{{HARNESS_ID}}" {{PLUGIN_DATA_ARG}} doctor +``` + +Summarize, using only fields the JSON actually reports: whether capture is +active or paused, the user ID and repository scope (`repo_id`), whether an +API key is configured, the event/flush/retrieval counts (`flushes` is the +number of completed flushes, not a pending count), and the doctor check +results. If doctor reports an authentication failure (401 / invalid key), say +clearly that the Mem0 API key is invalid or expired and that memories are NOT +being created. Never report an auth failure as "no memories found". Suggest +reinstalling with `--config api_key=...` in that case. diff --git a/integrations/agent-plugin-core/tests/test_build.py b/integrations/agent-plugin-core/tests/test_build.py new file mode 100644 index 000000000..dfa154f3d --- /dev/null +++ b/integrations/agent-plugin-core/tests/test_build.py @@ -0,0 +1,116 @@ +from __future__ import annotations + +import sys +import json +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +REPOSITORY_ROOT = ROOT.parents[1] +sys.path.insert(0, str(ROOT)) + +from build.build import build, bundle_drift, render_template, replace_output # noqa: E402 +from build.validate import validate_bundle # noqa: E402 + + +def test_render_rejects_unknown_or_unresolved_tokens() -> None: + with pytest.raises(ValueError, match="UNKNOWN"): + render_template("run {{UNKNOWN}}", {}) + + +def test_build_replaces_only_the_requested_output(tmp_path: Path) -> None: + staged = tmp_path / "staged" + staged.mkdir() + (staged / "plugin.json").write_text("{}", encoding="utf-8") + output = tmp_path / "output" + output.mkdir() + (output / "stale.py").write_text("stale", encoding="utf-8") + sibling = tmp_path / "keep.txt" + sibling.write_text("keep", encoding="utf-8") + + replace_output(staged, output) + + assert not (output / "stale.py").exists() + assert (output / "plugin.json").exists() + assert sibling.read_text(encoding="utf-8") == "keep" + + +def test_build_cannot_replace_an_installable_source_directory(tmp_path: Path) -> None: + staged = tmp_path / "staged" + staged.mkdir() + + with pytest.raises(ValueError, match="protected output path"): + replace_output(staged, REPOSITORY_ROOT / "integrations" / "claude-code-plugin") + + +def test_portable_bundle_is_conformant_and_self_contained(tmp_path: Path) -> None: + root = build("mem0-agent-plugin", "portable", tmp_path / "mem0-agent-plugin") + + assert validate_bundle(root, "portable") == [] + assert json.loads((root / "plugin.json").read_text(encoding="utf-8"))["$schema"] == ( + "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json" + ) + server = json.loads((root / "mcp.json").read_text(encoding="utf-8"))["mcpServers"]["mem0"] + assert server["type"] == "stdio" + assert server["args"] == ["${PLUGIN_ROOT}/core/mcp_server.py"] + assert "env" not in server + assert not (root / "core" / "hook_runner.py").exists() + assert not (root / "core" / "flush_worker.py").exists() + for skill in (root / "skills").glob("*/SKILL.md"): + frontmatter = skill.read_text(encoding="utf-8").split("---", 2)[1] + keys = {line.split(":", 1)[0] for line in frontmatter.splitlines() if ":" in line} + assert keys <= {"name", "description", "license", "compatibility", "metadata", "allowed-tools"} + assert not any(path.is_symlink() for path in root.rglob("*")) + + +@pytest.mark.parametrize("host", ["claude-code", "cursor", "codex", "kimi", "antigravity"]) +def test_native_bundle_is_self_contained(host: str, tmp_path: Path) -> None: + root = build(host, "native", tmp_path / host) + + assert (root / "core" / "memory_core.py").is_file() + assert (root / "skills" / "remember" / "SKILL.md").is_file() + assert not any(path.is_symlink() for path in root.rglob("*")) + + +@pytest.mark.parametrize("host", ["claude-code", "cursor", "codex", "kimi", "antigravity"]) +def test_native_control_skills_select_the_host_store(host: str, tmp_path: Path) -> None: + root = build(host, "native", tmp_path / host) + status = (root / "skills" / "status" / "SKILL.md").read_text(encoding="utf-8") + + assert f'--harness "{host}"' in status + if host == "claude-code": + assert '--plugin-data-dir "${CLAUDE_PLUGIN_DATA}"' in status + elif host == "codex": + assert '--plugin-data-dir "${PLUGIN_DATA}"' in status + + +@pytest.mark.parametrize( + ("host", "kind"), + [ + ("mem0-agent-plugin", "portable"), + ("claude-code", "native"), + ("cursor", "native"), + ("codex", "native"), + ("kimi", "native"), + ("antigravity", "native"), + ], +) +def test_installable_plugin_directories_are_current(host: str, kind: str) -> None: + assert bundle_drift(host, kind) == [] + + +def test_marketplaces_keep_public_names_and_reference_real_plugins() -> None: + marketplace = json.loads((REPOSITORY_ROOT / "marketplace.json").read_text(encoding="utf-8")) + sources = {plugin["name"]: plugin["source"] for plugin in marketplace["plugins"]} + + assert sources == {"mem0": "./integrations/claude-code-plugin"} + for source in sources.values(): + assert (REPOSITORY_ROOT / source).exists() + + codex_marketplace = json.loads( + (REPOSITORY_ROOT / ".agents" / "plugins" / "marketplace.json").read_text(encoding="utf-8") + ) + assert [plugin["name"] for plugin in codex_marketplace["plugins"]] == ["mem0"] + codex = codex_marketplace["plugins"][0] + assert codex["source"]["path"] == "./integrations/codex-plugin" diff --git a/integrations/agent-plugin-core/tests/test_conformance.py b/integrations/agent-plugin-core/tests/test_conformance.py new file mode 100644 index 000000000..53c7fb826 --- /dev/null +++ b/integrations/agent-plugin-core/tests/test_conformance.py @@ -0,0 +1,190 @@ +from __future__ import annotations + +import json +import os +import subprocess +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from conformance import run as conformance_run # noqa: E402 +from conformance.run import _command_check # noqa: E402 + + +PLUGIN_ROOT = Path(__file__).resolve().parents[1] +RUNNER = PLUGIN_ROOT / "conformance" / "run.py" +PYTHON_HOSTS = {"claude-code", "cursor", "codex", "kimi", "antigravity"} + + +def test_python_bundle_conformance_builds_every_host(tmp_path: Path) -> None: + report = tmp_path / "report.json" + artifacts = tmp_path / "artifacts" + + result = subprocess.run( + [ + sys.executable, + str(RUNNER), + "--group", + "python-bundles", + "--artifacts-dir", + str(artifacts), + "--report", + str(report), + ], + cwd=PLUGIN_ROOT.parents[1], + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 0, result.stdout + result.stderr + payload = json.loads(report.read_text(encoding="utf-8")) + assert payload["status"] == "passed" + assert { + entry["host"] + for entry in payload["checks"] + if entry["kind"] == "native" + } == PYTHON_HOSTS + assert { + entry["host"] + for entry in payload["checks"] + if entry["kind"] == "portable" + } == {"mem0-agent-plugin"} + for host in PYTHON_HOSTS: + assert (artifacts / host).is_dir() + assert (artifacts / "mem0-agent-plugin").is_dir() + + +def test_conformance_plan_covers_every_runtime(tmp_path: Path) -> None: + report = tmp_path / "plan.json" + + result = subprocess.run( + [sys.executable, str(RUNNER), "--list", "--report", str(report)], + cwd=PLUGIN_ROOT.parents[1], + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 0, result.stdout + result.stderr + payload = json.loads(report.read_text(encoding="utf-8")) + assert {entry["group"] for entry in payload["checks"]} == { + "python-bundles", + "python-tests", + "typescript-core", + "openclaw", + "opencode", + "pi-agent", + "deepseek", + } + assert all(entry["status"] == "planned" for entry in payload["checks"]) + assert { + entry["group"] + for entry in payload["checks"] + if entry["name"].endswith("-artifact") + } == {"openclaw", "opencode", "pi-agent", "deepseek"} + + +def test_typescript_artifact_check_rejects_monorepo_imports(tmp_path: Path) -> None: + dist = tmp_path / "dist" + dist.mkdir() + (dist / "index.js").write_text( + 'import { createMemoryLifecycle } from "../../agent-plugin-core/typescript/src/lifecycle.ts";\n', + encoding="utf-8", + ) + + artifact_check = getattr(conformance_run, "_typescript_artifact_check", None) + assert artifact_check is not None, "TypeScript package artifacts are not checked" + result = artifact_check("example", tmp_path, ("dist/index.js",)) + + assert result["status"] == "failed" + assert "monorepo source import" in result["output"] + + +def test_live_conformance_requires_an_explicit_mem0_key(tmp_path: Path) -> None: + environment = dict(os.environ) + environment.pop("MEM0_API_KEY", None) + + result = subprocess.run( + [sys.executable, str(RUNNER), "--live", "--report", str(tmp_path / "report.json")], + cwd=PLUGIN_ROOT.parents[1], + env=environment, + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 2 + assert "MEM0_API_KEY is required for --live" in result.stderr + + +def test_live_conformance_reports_platform_failure_without_crashing(tmp_path: Path) -> None: + report = tmp_path / "report.json" + environment = { + **os.environ, + "MEM0_API_KEY": "m0-intentionally-invalid", + "MEM0_API_URL": "http://127.0.0.1:1", + } + + result = subprocess.run( + [ + sys.executable, + str(RUNNER), + "--group", + "live-platform", + "--live", + "--report", + str(report), + ], + cwd=PLUGIN_ROOT.parents[1], + env=environment, + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 1 + payload = json.loads(report.read_text(encoding="utf-8")) + assert payload["status"] == "failed" + assert payload["checks"][0]["group"] == "live-platform" + + +def test_runtime_checks_preserve_the_callers_telemetry_setting(tmp_path: Path, monkeypatch) -> None: + monkeypatch.delenv("MEM0_TELEMETRY", raising=False) + + result = _command_check( + "environment", + "test", + [sys.executable, "-c", "import os; print(os.environ.get('MEM0_TELEMETRY', 'unset'))"], + cwd=tmp_path, + ) + + assert result["status"] == "passed" + assert result["output"] == "unset" + + +def test_command_check_can_force_non_interactive_installs(tmp_path: Path) -> None: + result = _command_check( + "environment", + "test", + [sys.executable, "-c", "import os; print(os.environ['CI'])"], + cwd=tmp_path, + environment_overrides={"CI": "true"}, + ) + + assert result["status"] == "passed" + assert result["output"] == "true" + + +def test_conformance_report_redacts_command_output(tmp_path: Path) -> None: + secret = "sk-eval-12345678901234567890" + + result = _command_check( + "redaction", + "test", + [sys.executable, "-c", f"print('failure: {secret}')"], + cwd=tmp_path, + ) + + assert secret not in json.dumps(result) + assert "[REDACTED]" in result["output"] diff --git a/integrations/agent-plugin-core/tests/test_sidekick.py b/integrations/agent-plugin-core/tests/test_sidekick.py new file mode 100644 index 000000000..9503e946e --- /dev/null +++ b/integrations/agent-plugin-core/tests/test_sidekick.py @@ -0,0 +1,83 @@ +"""Exercise bundled subagent hooks as separate host processes, without Mem0 calls.""" + +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "python")) + +from memory_core import EvidenceStore # noqa: E402 + + +@pytest.mark.parametrize("host", ["claude-code", "codex", "kimi", "cursor"]) +def test_sidekick_hooks_preserve_parent_scope_and_correlate_completion(tmp_path, host): + parent = tmp_path / "parent" + child = tmp_path / "child-worktree" + parent.mkdir() + child.mkdir() + database = tmp_path / "data" / "evidence.sqlite3" + store = EvidenceStore(database) + repo = store.repo_for_session("parent-session", str(parent)) + store.mark_injected("parent-session", repo.identity, [{"id": "one", "memory": "Parent memory marker."}]) + store.mark_injected("other-session", repo.identity, [{"id": "two", "memory": "Foreign memory marker."}]) + store.close() + + plugin = ROOT.parent / f"{host}-plugin" + adapter = plugin / ("adapters/claude/hook.py" if host == "claude-code" else "hooks/adapter.py") + start, stop = "sidekick-start", "sidekick-stop" + payload = {"session_id": "parent-session", "cwd": str(child), "agent_id": "worker", "agent_type": "sidekick"} + response_key = "last_assistant_message" + if host == "kimi": + start, stop = "SubagentStart", "SubagentStop" + payload["agent_name"] = payload.pop("agent_type") + response_key = "response" + elif host == "cursor": + start, stop = "subagentStart", "subagentStop" + payload = { + "conversation_id": "parent-session", + "workspace_roots": [str(child)], + "subagent_id": "worker", + "subagent_type": "sidekick", + } + response_key = "summary" + + env = {key: value for key, value in os.environ.items() if not key.startswith(("MEM0_", "CLAUDE_PLUGIN_"))} + env.update(MEM0_CODE_DATA_DIR=str(database.parent), MEM0_TELEMETRY="false", MEM0_API_URL="http://127.0.0.1:1") + + def invoke(event, body): + result = subprocess.run( + [sys.executable, str(adapter), event], + input=json.dumps({**body, "hook_event_name": event}), + text=True, + capture_output=True, + env=env, + timeout=15, + check=True, + ) + return result.stdout + + output = invoke(start, payload) + if host == "cursor": + assert json.loads(output) == {"permission": "allow"} + else: + context = output if host == "kimi" else json.loads(output)["hookSpecificOutput"]["additionalContext"] + assert "Parent memory marker." in context + assert "Foreign memory marker." not in output + assert "Parent memory marker." not in invoke(start, payload) + invoke(stop, {**payload, response_key: "Finished the delegated task."}) + + store = EvidenceStore(database) + runs = store.conn.execute("SELECT * FROM sidekick_runs").fetchall() + store.close() + assert len(runs) == 1 + assert (runs[0]["repo_id"], runs[0]["session_id"], runs[0]["agent_id"]) == ( + repo.identity, "parent-session", "worker" + ) + assert runs[0]["stopped_at"] + assert runs[0]["final_message"] == "Finished the delegated task." + assert (runs[0]["context_chars"] > 0) == (host != "cursor") diff --git a/integrations/agent-plugin-core/tests/test_validate.py b/integrations/agent-plugin-core/tests/test_validate.py new file mode 100644 index 000000000..79df69f54 --- /dev/null +++ b/integrations/agent-plugin-core/tests/test_validate.py @@ -0,0 +1,115 @@ +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] +VALIDATE = ROOT / "build" / "validate.py" +PLUGIN_SCHEMA = "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json" + + +def run_validator(bundle: Path) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [sys.executable, str(VALIDATE), str(bundle), "--kind", "portable"], + capture_output=True, + check=False, + text=True, + ) + + +def write_manifest(bundle: Path, schema: str = PLUGIN_SCHEMA) -> None: + bundle.mkdir(parents=True, exist_ok=True) + (bundle / "plugin.json").write_text( + json.dumps({"$schema": schema, "name": "mem0"}), + encoding="utf-8", + ) + + +def test_rejects_wrong_agent_plugin_schema(tmp_path: Path) -> None: + bundle = tmp_path / "plugin" + write_manifest(bundle, "https://agent-plugins.org/v1.0.0/plugin.schema.json") + + result = run_validator(bundle) + + assert result.returncode == 1 + assert result.stderr == ( + "plugin.json.$schema: " + "'https://agent-plugins.org/schemas/1.0.0/plugin.schema.json' was expected\n" + ) + + +def test_rejects_symlink_outside_bundle(tmp_path: Path) -> None: + bundle = tmp_path / "plugin" + write_manifest(bundle) + outside = tmp_path / "outside" + outside.mkdir() + (bundle / "skills").symlink_to(outside, target_is_directory=True) + + result = run_validator(bundle) + + assert result.returncode == 1 + assert result.stderr == "skills: symlinks are not allowed in release bundles\n" + + +def test_accepts_minimal_portable_bundle(tmp_path: Path) -> None: + bundle = tmp_path / "plugin" + write_manifest(bundle) + + result = run_validator(bundle) + + assert result.returncode == 0 + assert result.stdout == f"Validated portable bundle: {bundle}\n" + assert result.stderr == "" + + +def test_rejects_nonconformant_agent_skill(tmp_path: Path) -> None: + bundle = tmp_path / "plugin" + write_manifest(bundle) + skill = bundle / "skills" / "Bad_Name" + skill.mkdir(parents=True) + (skill / "SKILL.md").write_text( + "---\nname: Bad_Name\ndescription: invalid\nunknown: true\n---\n", + encoding="utf-8", + ) + + result = run_validator(bundle) + + assert result.returncode == 1 + assert "skills/Bad_Name" in result.stderr + + +def run_native_validator(bundle: Path) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [sys.executable, str(VALIDATE), str(bundle), "--kind", "native"], + capture_output=True, + check=False, + text=True, + ) + + +def test_native_bundle_rejects_malformed_json(tmp_path: Path) -> None: + bundle = tmp_path / "plugin" + bundle.mkdir(parents=True) + (bundle / "plugin.json").write_text("{bad json", encoding="utf-8") + + result = run_native_validator(bundle) + + assert result.returncode == 1 + assert "plugin.json" in result.stderr + + +def test_native_bundle_accepts_valid_json(tmp_path: Path) -> None: + bundle = tmp_path / "plugin" + bundle.mkdir(parents=True) + (bundle / "plugin.json").write_text('{"name": "mem0"}', encoding="utf-8") + hooks = bundle / "hooks" + hooks.mkdir() + (hooks / "hooks.json").write_text('{"version": 1}', encoding="utf-8") + + result = run_native_validator(bundle) + + assert result.returncode == 0 + assert "native" in result.stdout diff --git a/integrations/agent-plugin-core/typescript/package.json b/integrations/agent-plugin-core/typescript/package.json new file mode 100644 index 000000000..b9794e88a --- /dev/null +++ b/integrations/agent-plugin-core/typescript/package.json @@ -0,0 +1,14 @@ +{ + "name": "@mem0/agent-plugin-core", + "version": "0.0.0", + "private": true, + "type": "module", + "scripts": { + "test": "node --test tests/*.test.ts", + "typecheck": "tsc --noEmit" + }, + "devDependencies": { + "@types/node": "^22.15.0", + "typescript": "^5.6.0" + } +} diff --git a/integrations/agent-plugin-core/typescript/pnpm-lock.yaml b/integrations/agent-plugin-core/typescript/pnpm-lock.yaml new file mode 100644 index 000000000..65e364ea3 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/pnpm-lock.yaml @@ -0,0 +1,39 @@ +lockfileVersion: '9.0' + +settings: + autoInstallPeers: true + excludeLinksFromLockfile: false + +importers: + + .: + devDependencies: + '@types/node': + specifier: ^22.15.0 + version: 22.20.1 + typescript: + specifier: ^5.6.0 + version: 5.9.3 + +packages: + + '@types/node@22.20.1': + resolution: {integrity: sha512-EANqOCF9QFyra+4pfxUcX9STKJpCLjMbObVzljIJomAWSnuSIEAvyzEU53GaajbXJEgdh0iEcPL+DGvpUd4k1Q==} + + typescript@5.9.3: + resolution: {integrity: sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==} + engines: {node: '>=14.17'} + hasBin: true + + undici-types@6.21.0: + resolution: {integrity: sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==} + +snapshots: + + '@types/node@22.20.1': + dependencies: + undici-types: 6.21.0 + + typescript@5.9.3: {} + + undici-types@6.21.0: {} diff --git a/integrations/agent-plugin-core/typescript/src/formatting.ts b/integrations/agent-plugin-core/typescript/src/formatting.ts new file mode 100644 index 000000000..8f569cd94 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/src/formatting.ts @@ -0,0 +1,70 @@ +export interface MemoryLike { + id: string; + memory?: string; + categories?: string[]; + createdAt?: Date | string; +} + +export const MAX_OUTPUT_LINES = 200; +export const MAX_OUTPUT_CHARS = 50_000; +export const MAX_OUTPUT_BYTES = MAX_OUTPUT_CHARS; + +export function formatAge(date: Date | string): string { + const minutes = Math.floor((Date.now() - new Date(date).getTime()) / 60_000); + if (minutes < 60) return `${minutes}m ago`; + const hours = Math.floor(minutes / 60); + return hours < 24 ? `${hours}h ago` : `${Math.floor(hours / 24)}d ago`; +} + +export function formatMemoryCompact(memory: MemoryLike): string { + const category = memory.categories?.[0] ?? "uncategorized"; + const age = memory.createdAt ? ` (${formatAge(memory.createdAt)})` : ""; + return `[${category}] ${memory.memory ?? "(empty)"}${age} [mem0:${memory.id}]`; +} + +export function formatMemoryList(memories: MemoryLike[]): string { + return memories.length + ? memories.map((memory, index) => `${index + 1}. ${formatMemoryCompact(memory)}`).join("\n") + : "No memories found."; +} + +export function formatAddResult(result: unknown): string { + const items: MemoryLike[] = Array.isArray(result) + ? result + : ((result as { results?: MemoryLike[] } | null)?.results ?? (result ? [result as MemoryLike] : [])); + const pending = items.find((item) => (item as { status?: string }).status === "PENDING") as + | { eventId?: string; event_id?: string } + | undefined; + if (pending) { + const id = pending.eventId ?? pending.event_id; + return `Memory queued for background extraction${id ? ` (event ${id})` : ""}; it will be searchable shortly.`; + } + if (!items.length) return "Memory stored."; + return `Stored ${items.length} ${items.length === 1 ? "memory" : "memories"}:\n${formatMemoryList(items)}`; +} + +export function groupByCategory(memories: MemoryLike[]): Map { + const groups = new Map(); + for (const memory of memories) { + const category = memory.categories?.[0] ?? "uncategorized"; + groups.set(category, [...(groups.get(category) ?? []), memory]); + } + return groups; +} + +export function truncateOutput( + text: string, + maxChars = MAX_OUTPUT_CHARS, + maxLines = MAX_OUTPUT_LINES, +): string { + const lines = text.split("\n"); + if (lines.length <= maxLines && text.length <= maxChars) return text; + const kept = lines.slice(0, maxLines); + let result = kept.join("\n"); + const charCapped = result.length > maxChars; + if (charCapped) result = result.slice(0, maxChars); + const reasons = []; + if (kept.length < lines.length) reasons.push(`showing ${kept.length} of ${lines.length} lines`); + if (charCapped) reasons.push(`cut at ${Math.floor(maxChars / 1000)}KB`); + return `${result}\n\n[Output truncated: ${reasons.join(", ")}]`; +} diff --git a/integrations/agent-plugin-core/typescript/src/identity.ts b/integrations/agent-plugin-core/typescript/src/identity.ts new file mode 100644 index 000000000..6fffdee93 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/src/identity.ts @@ -0,0 +1,33 @@ +export interface EntityParams { + userId?: string; + agentId?: string; + runId?: string; +} + +const clean = (value: string | undefined): string | undefined => value?.trim() || undefined; + +export function entitySearchFilters( + params: EntityParams, + defaultUserId: string, +): Record { + const filters: Record = { user_id: clean(params.userId) ?? defaultUserId }; + const agentId = clean(params.agentId); + const runId = clean(params.runId); + if (agentId) filters.agent_id = agentId; + if (runId) filters.run_id = runId; + return filters; +} + +export function entityAddParams(params: EntityParams, defaultUserId: string): Record { + const values: Record = { userId: clean(params.userId) ?? defaultUserId }; + const agentId = clean(params.agentId); + const runId = clean(params.runId); + if (agentId) values.agentId = agentId; + if (runId) values.runId = runId; + return values; +} + +export function parseProjectFromRemote(remote: string): string | null { + const match = remote.trim().match(/[:/]([^/:]+)\/([^/:]+?)(?:\.git)?\/?$/); + return match ? `${match[1]}-${match[2]}` : null; +} diff --git a/integrations/agent-plugin-core/typescript/src/lifecycle.ts b/integrations/agent-plugin-core/typescript/src/lifecycle.ts new file mode 100644 index 000000000..42c522b54 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/src/lifecycle.ts @@ -0,0 +1,177 @@ +import type { MemoryLike } from "./formatting.ts"; +import { formatMemoryCompact } from "./formatting.ts"; + +const MAX_RECALL_QUERY_CHARS = 6_000; +export const DEFAULT_MAX_CONTEXT_CHARS = 4_000; + +const SECRET_PATTERNS: Array<[RegExp, string]> = [ + [/(authorization\s*[:=]\s*(?:bearer|token)\s+)[^\s"']+/gi, "$1[REDACTED]"], + [ + /((?:api[_-]?key|secret[_-]?access[_-]?key|session[_-]?token)\s*[:=]\s*)[^\s"']+/gi, + "$1[REDACTED]", + ], + [ + /((?:access[_-]?token|refresh[_-]?token|password|credential)\s*[:=]\s*)[^\s&"']+/gi, + "$1[REDACTED]", + ], + [/\b(?:sk|m0|mem0_sk|psk)-[A-Za-z0-9_-]{12,}\b/g, "[REDACTED]"], + [/\b(?:ASIA|AKIA)[A-Z0-9]{12,}\b/g, "[REDACTED]"], + [/\b(?:ghp_|github_pat_|xox[baprs]-)[A-Za-z0-9_-]{12,}\b/g, "[REDACTED]"], + [/-----BEGIN [^-]*PRIVATE KEY-----[\s\S]*?-----END [^-]*PRIVATE KEY-----/g, "[REDACTED]"], +]; + +export function redactSecrets(value: unknown): string { + let text = + typeof value === "string" + ? value + : (JSON.stringify(value, null, 0) ?? String(value)); + for (const [pattern, replacement] of SECRET_PATTERNS) text = text.replace(pattern, replacement); + return text; +} + +export function boundedText(value: unknown, limit: number): string { + const text = redactSecrets(value).trim(); + return text.length <= limit ? text : `${text.slice(0, limit)}\n...[truncated ${text.length - limit} chars]`; +} + +interface MessageLike { + role: string; + content?: unknown; +} + +function extractText(content: unknown): string | null { + if (typeof content === "string") return content; + if (!Array.isArray(content)) return null; + const text = content + .filter( + (block): block is { type: "text"; text: string } => + typeof block === "object" && + block !== null && + (block as { type?: unknown }).type === "text" && + typeof (block as { text?: unknown }).text === "string", + ) + .map((block) => block.text) + .join("\n"); + return text || null; +} + +export function extractConversation( + messages: MessageLike[], +): Array<{ role: "user" | "assistant"; content: string }> { + const conversation: Array<{ role: "user" | "assistant"; content: string }> = []; + for (const message of messages) { + if (message.role !== "user" && message.role !== "assistant") continue; + const text = extractText(message.content); + if (!text) continue; + const content = redactSecrets(text).trim(); + if (content) conversation.push({ role: message.role, content }); + } + return conversation; +} + +interface RecallOptions { + maxChars?: number; + seenIds?: Set; + timeoutMs?: number; +} + +interface MemoryLifecycleOptions { + maxContextChars?: number; + recallTimeoutMs?: number; +} + +/** Shared lifecycle policy. Host adapters only translate native events into these operations. */ +class MemoryLifecycle { + readonly #seenMemoryIds = new Set(); + readonly #options: MemoryLifecycleOptions; + + constructor(options: MemoryLifecycleOptions = {}) { + this.#options = options; + } + + beginSession(): void { + this.#seenMemoryIds.clear(); + } + + prepareConversation( + messages: MessageLike[], + ): Array<{ role: "user" | "assistant"; content: string }> { + return extractConversation(messages); + } + + prepareUserText(value: unknown): string { + return redactSecrets(value).trim(); + } + + recall( + prompt: string, + enabled: boolean, + search: (query: string) => Promise<{ results?: unknown[] }>, + ): Promise { + return buildRecallContext(prompt, enabled, search, { + maxChars: this.#options.maxContextChars, + seenIds: this.#seenMemoryIds, + timeoutMs: this.#options.recallTimeoutMs, + }); + } +} + +export function createMemoryLifecycle( + options: MemoryLifecycleOptions = {}, +): MemoryLifecycle { + return new MemoryLifecycle(options); +} + +export async function buildRecallContext( + prompt: string, + enabled: boolean, + search: (query: string) => Promise<{ results?: unknown[] }>, + options: RecallOptions = {}, +): Promise { + if (!enabled) return ""; + const query = boundedText(prompt, MAX_RECALL_QUERY_CHARS); + if (!query) return ""; + + try { + let timer: ReturnType | undefined; + let response: { results?: unknown[] } | null; + try { + const timeout = new Promise((resolve) => { + timer = setTimeout(() => resolve(null), options.timeoutMs ?? 2_000); + }); + response = await Promise.race([search(query), timeout]); + } finally { + if (timer) clearTimeout(timer); + } + if (!response) return ""; + const memories = (response.results ?? []) as MemoryLike[]; + const unseen = memories.filter((memory) => !options.seenIds?.has(memory.id)); + if (!unseen.length) return ""; + + const prefix = + "\nRetrieved automatically for the current request. This is a shallow first pass — search mem0_memory for more if you need it.\n"; + const suffix = "\n"; + const maxChars = options.maxChars ?? DEFAULT_MAX_CONTEXT_CHARS; + const lines: string[] = []; + for (const memory of unseen) { + const line = `${lines.length + 1}. ${redactSecrets(formatMemoryCompact(memory)) + .replace(/\s+/g, " ") + .trim()}`; + const candidate = prefix + [...lines, line].join("\n") + suffix; + if (candidate.length > maxChars) { + if (!lines.length) { + const available = maxChars - prefix.length - suffix.length; + if (available > 1) lines.push(`${line.slice(0, available - 1).trimEnd()}…`); + } + break; + } + lines.push(line); + options.seenIds?.add(memory.id); + } + if (!lines.length) return ""; + if (unseen[0] && !options.seenIds?.has(unseen[0].id)) options.seenIds?.add(unseen[0].id); + return prefix + lines.join("\n") + suffix; + } catch { + return ""; + } +} diff --git a/integrations/agent-plugin-core/typescript/src/scoping.ts b/integrations/agent-plugin-core/typescript/src/scoping.ts new file mode 100644 index 000000000..055ef2222 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/src/scoping.ts @@ -0,0 +1,50 @@ +export type Scope = "project" | "session" | "global"; + +export interface ScopeContext { + userId: string; + appId: string; + runId: string; +} + +export function normalizeScope(value: unknown): Scope { + return value === "session" || value === "global" ? value : "project"; +} + +export function resolveToolScope(requested: Scope | undefined, configured: Scope): Scope { + const scope = requested ?? configured; + if (scope === "global" && configured !== "global") { + throw new Error("Select global scope in the plugin settings or /mem0-scope command first."); + } + return scope; +} + +function validateContext(scope: Scope, context: ScopeContext): void { + const keys: (keyof ScopeContext)[] = ["userId"]; + if (scope !== "global") keys.push("appId"); + if (scope === "session") keys.push("runId"); + for (const key of keys) { + if (!context[key]?.trim() || /^\*+$/.test(context[key].trim())) { + throw new Error(`Invalid memory scope ${key}`); + } + } +} + +export function scopeSearchFilters(scope: Scope, context: ScopeContext): Record { + validateContext(scope, context); + if (scope === "session") { + return { user_id: context.userId, app_id: context.appId, run_id: context.runId }; + } + return scope === "global" + ? { user_id: context.userId } + : { user_id: context.userId, app_id: context.appId }; +} + +export function scopeAddParams(scope: Scope, context: ScopeContext): Record { + validateContext(scope, context); + if (scope === "session") { + return { userId: context.userId, appId: context.appId, runId: context.runId }; + } + return scope === "global" + ? { userId: context.userId } + : { userId: context.userId, appId: context.appId }; +} diff --git a/integrations/agent-plugin-core/typescript/src/telemetry.ts b/integrations/agent-plugin-core/typescript/src/telemetry.ts new file mode 100644 index 000000000..c6a7544ce --- /dev/null +++ b/integrations/agent-plugin-core/typescript/src/telemetry.ts @@ -0,0 +1,158 @@ +import { redactSecrets } from "./lifecycle.ts"; + +const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"; +const POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/"; +const OFF_VALUES = new Set(["false", "0", "no", "off"]); +const PRIVATE_KEYS = new Set([ + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +]); + +export interface TelemetryConfig { + host: string; + source: string; + version: string; + distinctId: string | (() => string | undefined); + delivery?: (batch: Record[]) => void | Promise; + flushThreshold?: number; + flushIntervalMs?: number; + maxQueueSize?: number; + commonProperties?: Record; + eventName?: (event: string) => string; + enabled?: () => boolean; +} + +export function isTelemetryEnabled(): boolean { + const value = process.env.MEM0_TELEMETRY; + return value === undefined || !OFF_VALUES.has(value.toLowerCase()); +} + +function safeValue(value: unknown): unknown { + if (typeof value === "string") return redactSecrets(value); + if (Array.isArray(value)) return value.map(safeValue); + if (value && typeof value === "object") { + return Object.fromEntries( + Object.entries(value) + .filter(([key]) => !PRIVATE_KEYS.has(key.toLowerCase().replace(/[^a-z]/g, ""))) + .map(([key, nested]) => [key, safeValue(nested)]), + ); + } + return value; +} + +function safeProperties(properties: Record): Record { + return safeValue(properties) as Record; +} + +export function errorKind(error: unknown): string { + const text = (error instanceof Error ? error.message : String(error)).toLowerCase(); + if (text.includes("timeout") || text.includes("aborted")) return "timeout"; + if (text.includes("401") || text.includes("403") || text.includes("unauthor")) return "auth"; + if (text.includes("429") || text.includes("rate limit")) return "rate-limited"; + if (/50[0234]/.test(text)) return "server-error"; + if (text.includes("400") || text.includes("422")) return "bad-request"; + if (text.includes("fetch failed") || text.includes("enotfound")) return "network"; + return error instanceof Error ? error.constructor.name : "other"; +} + +export function createTelemetry(config: TelemetryConfig) { + let queue: Record[] = []; + let timer: ReturnType | undefined; + const flushThreshold = config.flushThreshold ?? 10; + const maxQueueSize = config.maxQueueSize ?? 100; + + const deliver = config.delivery ?? (async (batch: Record[]) => { + await fetch(POSTHOG_BATCH_URL, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ api_key: POSTHOG_API_KEY, batch }), + signal: AbortSignal.timeout(3_000), + }); + }); + + async function flush(): Promise { + if (!queue.length) return; + const batch = queue; + queue = []; + try { + await deliver(batch); + } catch { + // Telemetry must never affect plugin behavior. + } + } + + function beforeExit(): void { + void flush(); + } + + function build(event: string, properties: Record = {}): Record | null { + if (!(config.enabled?.() ?? isTelemetryEnabled())) return null; + try { + const distinctId = typeof config.distinctId === "function" ? config.distinctId() : config.distinctId; + if (!distinctId) return null; + return { + event: config.eventName?.(event) ?? event, + distinct_id: distinctId, + properties: { + ...safeProperties(properties), + ...safeProperties(config.commonProperties ?? {}), + host: config.host, + source: config.source, + language: "node", + plugin_version: config.version, + node_version: process.version, + os: process.platform, + $process_person_profile: false, + $lib: "posthog-node", + }, + }; + } catch { + return null; + } + } + + function capture(event: string, properties: Record = {}): void { + try { + const payload = build(event, properties); + if (!payload) return; + queue.push(payload); + if (queue.length > maxQueueSize) queue = queue.slice(-maxQueueSize); + if (!timer) { + timer = setInterval(() => void flush(), config.flushIntervalMs ?? 5_000); + timer.unref?.(); + process.on("beforeExit", beforeExit); + } + if (queue.length >= flushThreshold) void flush(); + } catch { + // Telemetry must never affect plugin behavior. + } + } + + function resetForTesting(): void { + queue = []; + if (timer) clearInterval(timer); + timer = undefined; + process.off("beforeExit", beforeExit); + } + + return { build, capture, flush, resetForTesting, queueForTesting: () => queue }; +} diff --git a/integrations/agent-plugin-core/typescript/tests/formatting.test.ts b/integrations/agent-plugin-core/typescript/tests/formatting.test.ts new file mode 100644 index 000000000..d96a2cf2c --- /dev/null +++ b/integrations/agent-plugin-core/typescript/tests/formatting.test.ts @@ -0,0 +1,39 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + MAX_OUTPUT_CHARS, + MAX_OUTPUT_LINES, + formatAddResult, + formatAge, + formatMemoryCompact, + formatMemoryList, + truncateOutput, +} from "../src/formatting.ts"; + +test("memory formatting preserves the existing compact host output", () => { + const now = Date.now(); + assert.equal(formatAge(new Date(now - 30 * 60_000)), "30m ago"); + assert.match( + formatMemoryCompact({ id: "abc", memory: "Dark mode", categories: ["preference"] }), + /^\[preference\] Dark mode \[mem0:abc\]$/, + ); + assert.equal(formatMemoryList([]), "No memories found."); + assert.match(formatMemoryList([{ id: "1", memory: "A" }, { id: "2", memory: "B" }]), /^1\..*\n2\./); +}); + +test("write formatting handles pending, stored, and empty results", () => { + assert.equal( + formatAddResult({ eventId: "evt-9", status: "PENDING" }), + "Memory queued for background extraction (event evt-9); it will be searchable shortly.", + ); + assert.match(formatAddResult([{ id: "1" }, { id: "2" }]), /^Stored 2 memories:/); + assert.equal(formatAddResult([]), "Memory stored."); +}); + +test("output truncation preserves small output and bounds large output", () => { + assert.equal(truncateOutput("a\nb"), "a\nb"); + const many = Array.from({ length: MAX_OUTPUT_LINES + 1 }, (_, index) => `line ${index}`).join("\n"); + assert.match(truncateOutput(many), /showing 200 of 201 lines/); + assert.match(truncateOutput("x".repeat(MAX_OUTPUT_CHARS + 1)), /cut at 50KB/); +}); diff --git a/integrations/agent-plugin-core/typescript/tests/identity.test.ts b/integrations/agent-plugin-core/typescript/tests/identity.test.ts new file mode 100644 index 000000000..c789f766f --- /dev/null +++ b/integrations/agent-plugin-core/typescript/tests/identity.test.ts @@ -0,0 +1,18 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { entityAddParams, entitySearchFilters, parseProjectFromRemote } from "../src/identity.ts"; + +test("parses common git remote forms", () => { + assert.equal(parseProjectFromRemote("git@github.com-work:mem0ai/mem0.git"), "mem0ai-mem0"); + assert.equal(parseProjectFromRemote("https://github.com/mem0ai/mem0/"), "mem0ai-mem0"); + assert.equal(parseProjectFromRemote("not-a-remote"), null); + assert.equal(parseProjectFromRemote(""), null); +}); + +test("entity filters trim overrides and preserve API casing", () => { + const params = { userId: " alice ", agentId: " agent ", runId: " " }; + assert.deepEqual(entitySearchFilters(params, "default"), { user_id: "alice", agent_id: "agent" }); + assert.deepEqual(entityAddParams(params, "default"), { userId: "alice", agentId: "agent" }); + assert.deepEqual(entitySearchFilters({ userId: " " }, "default"), { user_id: "default" }); +}); diff --git a/integrations/agent-plugin-core/typescript/tests/lifecycle.test.ts b/integrations/agent-plugin-core/typescript/tests/lifecycle.test.ts new file mode 100644 index 000000000..7e1776eae --- /dev/null +++ b/integrations/agent-plugin-core/typescript/tests/lifecycle.test.ts @@ -0,0 +1,120 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { + boundedText, + buildRecallContext, + createMemoryLifecycle, + extractConversation, + redactSecrets, +} from "../src/lifecycle.ts"; + +test("one lifecycle owns recall state and resets it for a new session", async () => { + const lifecycle = createMemoryLifecycle({ recallTimeoutMs: 50 }); + const search = async () => ({ results: [{ id: "m1", memory: "Use pnpm" }] }); + + assert.match(await lifecycle.recall("package manager", true, search), /Use pnpm/); + assert.equal(await lifecycle.recall("package manager", true, search), ""); + + lifecycle.beginSession(); + assert.match(await lifecycle.recall("package manager", true, search), /Use pnpm/); +}); + +test("one lifecycle owns capture preparation", () => { + const lifecycle = createMemoryLifecycle(); + + assert.deepEqual( + lifecycle.prepareConversation([ + { role: "user", content: "password=secret-value" }, + { role: "assistant", content: "Configured it" }, + ]), + [ + { role: "user", content: "password=[REDACTED]" }, + { role: "assistant", content: "Configured it" }, + ], + ); +}); + +test("capture preserves long prompts and responses while redacting secrets", () => { + const lifecycle = createMemoryLifecycle(); + const prompt = "Repository question. ".repeat(2000) + "Final requirement. api_key=hidden-user-secret"; + const answer = "Repository answer. ".repeat(4000) + "Final detail. password=hidden-agent-secret"; + assert.deepEqual(lifecycle.prepareConversation([ + { role: "user", content: prompt }, + { role: "assistant", content: [{ type: "text", text: answer }] }, + ]), [ + { role: "user", content: redactSecrets(prompt) }, + { role: "assistant", content: redactSecrets(answer) }, + ]); + assert.equal(lifecycle.prepareUserText(prompt), redactSecrets(prompt)); +}); + +test("redacts Claude-equivalent credentials before content leaves the host", () => { + const privateKey = "-----BEGIN PRIVATE KEY-----\nsecret\n-----END PRIVATE KEY-----"; + const input = [ + "Authorization: Bearer top-secret-token", + "api_key=super-secret-value", + "password=hunter2", + "ghp_abcdefghijklmnopqrstuvwxyz123456", + privateKey, + ].join("\n"); + + const output = redactSecrets(input); + assert.equal(output.includes("top-secret-token"), false); + assert.equal(output.includes("super-secret-value"), false); + assert.equal(output.includes("hunter2"), false); + assert.equal(output.includes("ghp_"), false); + assert.equal(output.includes("secret\n-----END"), false); + assert.match(output, /\[REDACTED\]/); +}); + +test("bounds redacted content and reports the omitted character count", () => { + assert.equal(boundedText(" abc ", 10), "abc"); + assert.equal(boundedText("abcdefgh", 5), "abcde\n...[truncated 3 chars]"); +}); + +test("normalizes and sanitizes user/assistant conversation content", () => { + const messages = [ + { role: "system", content: "ignored" }, + { role: "user", content: [{ type: "text", text: "token=m0-abcdefghijklmnop" }] }, + { role: "assistant", content: [{ type: "tool_use" }, { type: "text", text: "done" }] }, + ]; + + assert.deepEqual(extractConversation(messages), [ + { role: "user", content: "token=[REDACTED]" }, + { role: "assistant", content: "done" }, + ]); +}); + +test("recall is bounded, fail-open, and de-duplicates already injected memories", async () => { + const seen = new Set(["old"]); + const search = async () => ({ + results: [ + { id: "old", memory: "already shown" }, + { id: "new", memory: `api_key=hidden ${"x".repeat(100)}` }, + ], + }); + + const output = await buildRecallContext("what changed?", true, search, { + maxChars: 240, + seenIds: seen, + }); + assert.equal(output.includes("already shown"), false); + assert.equal(output.includes("hidden"), false); + assert.ok(output.length <= 240); + assert.equal(seen.has("new"), true); + assert.equal( + await buildRecallContext("what changed?", true, async () => { + throw new Error("offline"); + }), + "", + ); +}); + +test("recall times out without blocking the host turn", async () => { + const never = () => new Promise<{ results?: unknown[] }>(() => {}); + const started = Date.now(); + + assert.equal(await buildRecallContext("hello", true, never, { timeoutMs: 5 }), ""); + assert.ok(Date.now() - started < 100); +}); diff --git a/integrations/agent-plugin-core/typescript/tests/scoping.test.ts b/integrations/agent-plugin-core/typescript/tests/scoping.test.ts new file mode 100644 index 000000000..266457649 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/tests/scoping.test.ts @@ -0,0 +1,44 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { normalizeScope, resolveToolScope, scopeAddParams, scopeSearchFilters } from "../src/scoping.ts"; + +const context = { userId: "u", appId: "app", runId: "run" }; + +test("normalizes unknown scope to project", () => { + assert.equal(normalizeScope("session"), "session"); + assert.equal(normalizeScope("global"), "global"); + assert.equal(normalizeScope("invalid"), "project"); +}); + +test("resolves project, session, and global search filters", () => { + assert.deepEqual(scopeSearchFilters("project", context), { user_id: "u", app_id: "app" }); + assert.deepEqual(scopeSearchFilters("session", context), { user_id: "u", app_id: "app", run_id: "run" }); + assert.deepEqual(scopeSearchFilters("global", context), { user_id: "u" }); +}); + +test("resolves camel-case add params", () => { + assert.deepEqual(scopeAddParams("project", context), { userId: "u", appId: "app" }); + assert.deepEqual(scopeAddParams("session", context), { userId: "u", appId: "app", runId: "run" }); + assert.deepEqual(scopeAddParams("global", context), { userId: "u" }); +}); + +test("rejects empty and wildcard identities before building filters or writes", () => { + for (const invalid of ["", " ", "*", "***"]) { + for (const scope of ["project", "session"] as const) { + for (const resolve of [scopeSearchFilters, scopeAddParams]) { + assert.throws(() => resolve(scope, { ...context, appId: invalid }), /appId/); + assert.throws(() => resolve(scope, { ...context, userId: invalid }), /userId/); + } + } + assert.throws(() => scopeSearchFilters("session", { ...context, runId: invalid }), /runId/); + } +}); + + +test("tools cannot enable global scope without a user-configured global default", () => { + assert.throws(() => resolveToolScope("global", "project"), /Select global/); + assert.throws(() => resolveToolScope("global", "session"), /Select global/); + assert.equal(resolveToolScope("global", "global"), "global"); + assert.equal(resolveToolScope("session", "project"), "session"); +}); diff --git a/integrations/agent-plugin-core/typescript/tests/telemetry.test.ts b/integrations/agent-plugin-core/typescript/tests/telemetry.test.ts new file mode 100644 index 000000000..e04794614 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/tests/telemetry.test.ts @@ -0,0 +1,124 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { createTelemetry, errorKind } from "../src/telemetry.ts"; + +test("telemetry preserves host names and strips sensitive properties", async () => { + const delivered: Record[][] = []; + const telemetry = createTelemetry({ + host: "deepseek", + source: "DEEPSEEK_HARNESS", + version: "1.2.3", + distinctId: "person", + delivery: async (batch) => { + delivered.push(batch); + }, + }); + + telemetry.capture("deepseek.tool.search_memory", { + success: true, + query: "secret", + apiKey: "key", + cwd: "/private/repo", + repo_id: "raw-repo", + query_chars: 6, + }); + await telemetry.flush(); + telemetry.resetForTesting(); + + const event = delivered[0][0] as { event: string; properties: Record }; + assert.equal(event.event, "deepseek.tool.search_memory"); + assert.deepEqual(event.properties, { + success: true, + query_chars: 6, + host: "deepseek", + source: "DEEPSEEK_HARNESS", + language: "node", + plugin_version: "1.2.3", + node_version: process.version, + os: process.platform, + $process_person_profile: false, + $lib: "posthog-node", + }); +}); + +test("all false-like opt-out values suppress events", async () => { + const original = process.env.MEM0_TELEMETRY; + try { + for (const value of ["false", "0", "no", "OFF"]) { + process.env.MEM0_TELEMETRY = value; + let delivered = false; + const telemetry = createTelemetry({ + host: "pi", + source: "PI_AGENT_PLUGIN", + version: "1", + distinctId: "person", + delivery: async () => { + delivered = true; + }, + }); + telemetry.capture("pi.test"); + await telemetry.flush(); + telemetry.resetForTesting(); + assert.equal(delivered, false); + } + } finally { + if (original === undefined) delete process.env.MEM0_TELEMETRY; + else process.env.MEM0_TELEMETRY = original; + } +}); + +test("telemetry redacts secrets nested inside allowed properties", () => { + const secret = "sk-eval-12345678901234567890"; + const telemetry = createTelemetry({ + host: "pi", + source: "PI_AGENT_PLUGIN", + version: "1", + distinctId: "person", + }); + + const event = telemetry.build("pi.test", { + note: `failure contained ${secret}`, + details: { authorization: `Bearer ${secret}`, count: 2 }, + }); + telemetry.resetForTesting(); + + const serialized = JSON.stringify(event); + assert.equal(serialized.includes(secret), false); + assert.equal(serialized.includes("[REDACTED]"), true); + assert.equal((event?.properties as { details: { count: number } }).details.count, 2); +}); + +for (const key of ["password", "token", "secret", "authorization"]) { + test(`telemetry removes ${key} from nested list elements`, () => { + const secret = "sk-eval-12345678901234567890"; + const telemetry = createTelemetry({ + host: "pi", + source: "PI_AGENT_PLUGIN", + version: "1", + distinctId: "person", + }); + + const event = telemetry.build("pi.test", { + details: [ + { [key]: "plain-value", count: 2 }, + { nested: { [key.toUpperCase()]: "plain-value", ok: true } }, + [`failure contained ${secret}`], + ], + }); + telemetry.resetForTesting(); + + assert.deepEqual((event?.properties as { details: unknown[] }).details, [ + { count: 2 }, + { nested: { ok: true } }, + ["failure contained [REDACTED]"], + ]); + }); +} + +test("error classification does not expose messages", () => { + assert.equal(errorKind(new Error("429 secret query")), "rate-limited"); + assert.equal(errorKind(new Error("401 key")), "auth"); + assert.equal(errorKind(new Error("request timeout")), "timeout"); + assert.equal(errorKind(new Error("fetch failed")), "network"); +}); diff --git a/integrations/agent-plugin-core/typescript/tsconfig.json b/integrations/agent-plugin-core/typescript/tsconfig.json new file mode 100644 index 000000000..5abe9f948 --- /dev/null +++ b/integrations/agent-plugin-core/typescript/tsconfig.json @@ -0,0 +1,12 @@ +{ + "compilerOptions": { + "target": "ES2022", + "module": "ES2022", + "moduleResolution": "bundler", + "strict": true, + "types": ["node"], + "allowImportingTsExtensions": true, + "noEmit": true + }, + "include": ["src", "tests"] +} diff --git a/integrations/antigravity-plugin/agents/sidekick/agent.md b/integrations/antigravity-plugin/agents/sidekick/agent.md new file mode 100644 index 000000000..6a4d8f29a --- /dev/null +++ b/integrations/antigravity-plugin/agents/sidekick/agent.md @@ -0,0 +1,17 @@ +--- +name: sidekick +description: Coding subagent for focused implementation, investigation, testing, debugging, or review work. +subagent: true +--- + +You are Mem0's coding sidekick. Complete only the bounded task the main agent +delegates to you and return a concise, self-contained result. + +Search Mem0 before work that may depend on prior repository decisions or user +preferences. Inspect the relevant repository rules and code, make changes when +asked, and run the smallest decisive validation. Use only the workspace the +caller assigned; do not assume a separate Git worktree. + +Your final response must state the outcome, changed files, validation, and any +remaining risk. Do not commit, push, or open a pull request unless the caller +explicitly asks. diff --git a/integrations/antigravity-plugin/core/flush_worker.py b/integrations/antigravity-plugin/core/flush_worker.py new file mode 100644 index 000000000..6f7b9ecb2 --- /dev/null +++ b/integrations/antigravity-plugin/core/flush_worker.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Detached remote checkpoint worker. + +Claude Code may cancel SessionEnd hooks as a print-mode process exits. The hook +therefore persists its input first and launches this process in a new session. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + checkpoint_session, + configure_harness, + touch_handoff_heartbeat, +) + + +def main() -> int: + if len(sys.argv) != 2: + return 2 + handoff_path = Path(sys.argv[1]) + os.environ["MEM0_CODE_HANDOFF_PATH"] = str(handoff_path) + harness = os.environ.get("MEM0_PLUGIN_HARNESS") + if harness: + source_tag = os.environ.get("MEM0_PLUGIN_SOURCE_TAG", "") + configure_harness( + harness, + env_prefix=os.environ.get("MEM0_PLUGIN_ENV_PREFIX", ""), + data_dir_name=os.environ.get("MEM0_PLUGIN_DATA_DIR_NAME", ""), + source_tag=source_tag, + ) + telemetry.init(harness=harness, source_tag=source_tag.upper()) + completed = False + try: + payload = json.loads(handoff_path.read_text(encoding="utf-8")) + delay = float(payload.get("delay_seconds") or 0) + if delay > 0: + payload.pop("delay_seconds", None) + temporary = handoff_path.with_suffix(f".{os.getpid()}.tmp") + try: + temporary.write_text(json.dumps(payload), encoding="utf-8") + temporary.replace(handoff_path) + finally: + temporary.unlink(missing_ok=True) + time.sleep(delay) + if not handoff_path.exists(): + return 0 + hook_input = payload.get("hook_input") or {} + reason = str(payload.get("reason") or "checkpoint") + wait_for_inflight = bool(payload.get("wait_for_inflight")) + store = EvidenceStore() + try: + if wait_for_inflight: + session_id = str( + hook_input.get("session_id") or "unknown-session" + ) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + deadline = time.monotonic() + float( + os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120") + ) + while ( + store.has_inflight_flush(repo.identity, session_id) + and time.monotonic() < deadline + ): + touch_handoff_heartbeat() + time.sleep(0.25) + # Hooks capture the conversation before handoff; the worker only flushes it. + result = checkpoint_session(store, hook_input, reason) + print(json.dumps(result, sort_keys=True), flush=True) + completed = result.get("status") in { + "semantic-succeeded", + "explicitly-stored", + "nothing-to-flush", + } + finally: + store.close() + return 0 + finally: + telemetry.flush() + if completed: + try: + handoff_path.unlink() + except OSError: + pass + elif handoff_path.suffix == ".running": + try: + handoff_path.replace(handoff_path.with_suffix(".json")) + except OSError: + pass + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/antigravity-plugin/core/hook_runner.py b/integrations/antigravity-plugin/core/hook_runner.py new file mode 100644 index 000000000..2a2adb557 --- /dev/null +++ b/integrations/antigravity-plugin/core/hook_runner.py @@ -0,0 +1,372 @@ +"""Shared hook orchestration for all Mem0 agent plugins.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import subprocess +import sys +import time +import uuid +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + _session_id, + api_key, + bounded, + cache_plugin_api_key, + checkpoint_session, + clear_stale_api_key_cache, + configure_harness, + data_dir, + detached_process_kwargs, + format_context, + harness_config, + record_session_start, + record_tool, + record_user_prompt, + redact, + search_memories, +) + +STALE_RUNNING_SECONDS = 300 +PENDING_EXPIRY_SECONDS = 7 * 24 * 60 * 60 +PENDING_LAUNCH_LIMIT = 5 +DEFAULT_IDLE_FLUSH_SECONDS = 300 + +_core_dir: Path = Path(__file__).resolve().parent + + +def read_hook_input() -> dict: + try: + value = json.load(sys.stdin) + return value if isinstance(value, dict) else {} + except (json.JSONDecodeError, OSError): + return {} + + +def default_record_stop(store: EvidenceStore, hook_input: dict): + """Record the assistant's response without transcript parsing.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + message = redact(hook_input.get("last_assistant_message", "")).strip() + if message: + store.record_assistant_response(repo, session_id, message) + return repo, session_id + + +def first_prompt_memory_output(store: EvidenceStore, hook_input: dict) -> dict: + """Search once before the agent handles the first prompt in a session.""" + repo, session_id, prompt, is_first_prompt = record_user_prompt(store, hook_input) + if not is_first_prompt: + return {} + try: + minimum_query_chars = int(os.environ.get("MEM0_CODE_MIN_QUERY_CHARS", "20")) + except ValueError: + minimum_query_chars = 20 + if len(prompt.strip()) < max(minimum_query_chars, 1): + return {} + result = search_memories( + store, repo, session_id, bounded(prompt, 6000), + top_k=5, operation="first-prompt-search", timeout=2, + ) + if not result.memories: + return {} + context = format_context( + result.memories, + "Mem0 found these relevant memories from earlier work in this repository:", + ) + telemetry.record( + "context_injected", + repo=repo, session_id=session_id, trigger="first-prompt", + memory_count=len(result.memories), context_chars=len(context), + prompt_chars=len(prompt), + ) + return { + "hookSpecificOutput": { + "hookEventName": "UserPromptSubmit", + "additionalContext": context, + }, + } + + +def _launch_handoff(handoff_path: Path) -> bool: + running_path = handoff_path.with_suffix(".running") + try: + handoff_path.replace(running_path) + except OSError: + return False + worker = _core_dir / "flush_worker.py" + log_path = data_dir() / "flush-worker.log" + log_handle = open(log_path, "a", encoding="utf-8") + harness = harness_config() + child_env = os.environ.copy() + child_env.update( + { + "MEM0_CODE_DATA_DIR": str(data_dir()), + "MEM0_PLUGIN_HARNESS": harness["name"], + "MEM0_PLUGIN_ENV_PREFIX": harness["env_prefix"], + "MEM0_PLUGIN_DATA_DIR_NAME": harness["data_dir_name"], + "MEM0_PLUGIN_SOURCE_TAG": harness["source_tag"], + } + ) + try: + subprocess.Popen( + [sys.executable, str(worker), str(running_path)], + stdin=subprocess.DEVNULL, + stdout=log_handle, stderr=log_handle, + close_fds=True, + env=child_env, + **detached_process_kwargs(), + ) + finally: + log_handle.close() + return True + + +def recover_pending_handoffs() -> int: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + now = time.time() + for running in pending_dir.glob("*.running"): + try: + if now - running.stat().st_mtime > STALE_RUNNING_SECONDS: + running.replace(running.with_suffix(".json")) + except OSError: + continue + recoverable = [] + for handoff in pending_dir.glob("*.json"): + try: + age = now - handoff.stat().st_mtime + except OSError: + continue + if age > PENDING_EXPIRY_SECONDS: + handoff.unlink(missing_ok=True) + continue + recoverable.append((age, handoff)) + recoverable.sort(key=lambda item: item[0], reverse=True) + launched = 0 + for _, handoff in recoverable[:PENDING_LAUNCH_LIMIT]: + launched += int(_launch_handoff(handoff)) + return launched + + +def refresh_pending_handoffs() -> None: + pending_dir = data_dir() / "pending" + if not pending_dir.is_dir(): + return + for pattern in ("*.json", "*.running"): + for handoff in pending_dir.glob(pattern): + try: + os.utime(handoff) + except OSError: + continue + + +def hand_off_flush( + hook_input: dict, reason: str, *, wait_for_inflight: bool = False, +) -> None: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = ( + f"{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}\0{reason}" + ) + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + handoff_path = pending_dir / f"{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": reason, + "wait_for_inflight": wait_for_inflight, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + + +def automatic_flush_enabled() -> bool: + return os.environ.get("MEM0_CODE_AUTO_FLUSH", "true").lower() in { + "1", "true", "yes", "on", + } + + +def schedule_periodic_checkpoint( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + if ( + not automatic_flush_enabled() + or not api_key() + or not store.checkpoint_due(repo.identity, session_id) + ): + return False + if store.prepare_flush(repo, session_id, "periodic") is None: + return False + hand_off_flush(hook_input, "periodic") + return True + + +def _idle_flush_seconds() -> int: + try: + return max( + int(os.environ.get("MEM0_CODE_IDLE_FLUSH_SECONDS", str(DEFAULT_IDLE_FLUSH_SECONDS))), + 0, + ) + except ValueError: + return DEFAULT_IDLE_FLUSH_SECONDS + + +def schedule_idle_flush( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + delay = _idle_flush_seconds() + if delay <= 0 or not automatic_flush_enabled() or not api_key(): + return False + if store.has_inflight_flush(repo.identity, session_id): + return False + if not store.has_unflushed_events(repo.identity, session_id): + return False + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = f"idle\0{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}" + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + for old in pending_dir.glob(f"idle-{digest}*"): + old.unlink(missing_ok=True) + handoff_path = pending_dir / f"idle-{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": "idle", + "delay_seconds": delay, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + return True + + +def log_failure(exc: Exception) -> None: + try: + log_path = data_dir() / "plugin-errors.log" + with log_path.open("a", encoding="utf-8") as handle: + handle.write(f"{time.time():.3f} {type(exc).__name__}: {exc}\n") + except OSError: + pass + + +def run( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> int: + if record_stop_fn is None: + record_stop_fn = default_record_stop + if automatic_flush_reasons is None: + automatic_flush_reasons = {"session-end"} + + base_actions = ["session-start", "user-prompt", "post-tool", "stop", "flush"] + all_actions = base_actions + list((extra_actions or {}).keys()) + + parser = argparse.ArgumentParser() + parser.add_argument("action", choices=all_actions) + parser.add_argument("--reason", default="manual") + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + args = parser.parse_args() + + if args.harness: + configure_harness(args.harness) + telemetry.init(harness=args.harness) + + if args.plugin_data_dir: + os.environ[data_dir_env] = args.plugin_data_dir + + cache_plugin_api_key() + if args.action == "session-start": + clear_stale_api_key_cache() + + hook_input = read_hook_input() + store = EvidenceStore() + try: + if store.is_paused(): + if args.action == "session-start": + refresh_pending_handoffs() + telemetry.record("session_start", paused=True) + telemetry.spawn_flush() + return 0 + + if args.action == "session-start": + if telemetry.is_first_run(): + telemetry.record("install") + recovered = recover_pending_handoffs() + record_session_start(store, hook_input) + if recovered: + telemetry.record("handoff_recovered", count=recovered) + telemetry.spawn_flush() + elif args.action == "user-prompt": + output = first_prompt_memory_output(store, hook_input) + if output: + print(json.dumps(output)) + elif args.action == "post-tool": + record_tool(store, hook_input) + elif args.action == "stop": + repo, session_id = record_stop_fn(store, hook_input) + if not schedule_periodic_checkpoint(store, hook_input, repo, session_id): + schedule_idle_flush(store, hook_input, repo, session_id) + elif args.action == "flush": + automatic = args.reason in automatic_flush_reasons + if automatic and not automatic_flush_enabled(): + return 0 + if args.reason == "session-end": + record_stop_fn(store, hook_input) + if os.environ.get("MEM0_CODE_SYNC_FLUSH") == "1": + print(json.dumps(checkpoint_session(store, hook_input, args.reason))) + else: + session_id = str(hook_input.get("session_id") or "unknown-session") + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + already_running = store.has_inflight_flush(repo.identity, session_id) + if already_running and args.reason == "session-end": + hand_off_flush(hook_input, args.reason, wait_for_inflight=True) + elif not already_running and store.prepare_flush( + repo, session_id, args.reason, + ) is not None: + hand_off_flush(hook_input, args.reason) + elif extra_actions and args.action in extra_actions: + result = extra_actions[args.action](store, hook_input) + if result: + print(json.dumps(result)) + finally: + store.close() + return 0 + + +def entry_point( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> None: + try: + raise SystemExit(run( + record_stop_fn=record_stop_fn, + extra_actions=extra_actions, + data_dir_env=data_dir_env, + automatic_flush_reasons=automatic_flush_reasons, + )) + except Exception as exc: + log_failure(exc) + raise SystemExit(0) + + +if __name__ == "__main__": + entry_point() diff --git a/integrations/antigravity-plugin/core/mcp_server.py b/integrations/antigravity-plugin/core/mcp_server.py new file mode 100644 index 000000000..036fbbdc9 --- /dev/null +++ b/integrations/antigravity-plugin/core/mcp_server.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Expose Mem0's memory search as one local coding-agent tool.""" + +from __future__ import annotations + +import json +import os +import sys +from typing import Any + +import telemetry +from memory_core import ( + CODING_MEMORY_CATEGORY_NAMES, + PLUGIN_VERSION, + SEARCH_SCOPES, + format_search_result, + resolve_repo, + search_memories, +) + +PROTOCOL_VERSION = "2024-11-05" +TOOL_NAME = "search_memories" +TOOL_DESCRIPTION = ( + "Search memories from earlier work in this repository. ALWAYS call this " + "tool before answering anything that could depend on prior context: the " + "user's preferences, facts about this codebase, history, people, projects, " + "or earlier decisions. Do not rely on the chat window alone. The " + "repository's memory is shared by everyone who works in it and includes " + "what it took to run, test, or build here, so search before assuming an " + "invocation works. The scope argument changes what is searched: 'repo' " + "(default) is the whole repository's shared memory plus your own " + "preferences, 'dir' narrows the shared part to the directory you are " + "working in, and 'mine' is your preferences alone." +) +TOOL_SCHEMA = { + "type": "object", + "properties": { + "query": { + "type": "string", + "minLength": 1, + "maxLength": 2000, + "description": "A direct question about earlier work in this repository.", + }, + "top_k": { + "type": "integer", + "minimum": 1, + "maximum": 20, + "description": "Maximum memories to return. Uses Mem0's configured default when omitted.", + }, + "category": { + "type": "string", + "enum": list(CODING_MEMORY_CATEGORY_NAMES), + "description": "Optional memory category. Omit to search every category.", + }, + "scope": { + "type": "string", + "enum": list(SEARCH_SCOPES), + "description": ( + "Which memories to search. 'repo' (default) is the whole repository's " + "shared memory plus your own preferences, 'dir' narrows the shared " + "part to the current directory, 'mine' is your preferences alone." + ), + }, + "run_id": { + "type": "string", + "minLength": 1, + "description": ( + "Optional coding-agent session ID. With any scope, restricts results to memories " + "saved in that session. Omit to recall memories across sessions." + ), + }, + }, + "required": ["query"], + "additionalProperties": False, +} + + +class ToolInputError(ValueError): + pass + + +def _validate_arguments( + arguments: Any, +) -> tuple[str, int | None, str | None, str | None, str | None]: + if not isinstance(arguments, dict): + raise ToolInputError("Search arguments must be an object.") + + unknown = set(arguments) - {"query", "top_k", "category", "scope", "run_id"} + if unknown: + raise ToolInputError(f"Unknown search argument: {sorted(unknown)[0]}") + + query = arguments.get("query") + if not isinstance(query, str) or not query.strip(): + raise ToolInputError("query must be a non-empty string.") + query = query.strip() + if len(query) > 2000: + raise ToolInputError("query must be at most 2,000 characters.") + + top_k = arguments.get("top_k") + if top_k is not None and ( + isinstance(top_k, bool) or not isinstance(top_k, int) or not 1 <= top_k <= 20 + ): + raise ToolInputError("top_k must be an integer from 1 to 20.") + + category = arguments.get("category") + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ToolInputError("category must be one of Mem0's supported categories.") + + scope = arguments.get("scope") + if scope is not None and scope not in SEARCH_SCOPES: + raise ToolInputError(f"scope must be one of {list(SEARCH_SCOPES)}.") + + run_id = arguments.get("run_id") + if run_id is not None: + if not isinstance(run_id, str) or not run_id.strip(): + raise ToolInputError("run_id must be a non-empty string.") + run_id = run_id.strip() + + return query, top_k, category, scope, run_id + + +def call_search_memories(arguments: Any, cwd: str | None = None) -> str: + query, top_k, category, scope, run_id = _validate_arguments(arguments) + repo = resolve_repo(cwd or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + result = search_memories( + None, + repo, + None, + query, + top_k=top_k, + category=category, + scope=scope, + run_id=run_id, + operation="mcp-search", + ) + return format_search_result(result) + + +def _workspace_cwd(params: dict[str, Any]) -> str | None: + meta = params.get("_meta") + if not isinstance(meta, dict): + return None + metadata = meta.get("x-codex-turn-metadata") + if not isinstance(metadata, dict): + return None + workspaces = metadata.get("workspaces") or {} + if isinstance(workspaces, dict): + return next((path for path in workspaces if isinstance(path, str) and path), None) + return None + + +def _tool_response(text: str, *, is_error: bool = False) -> dict[str, Any]: + return { + "content": [{"type": "text", "text": text}], + "isError": is_error, + } + + +def handle_request(message: Any) -> dict[str, Any] | None: + if not isinstance(message, dict): + return None + request_id = message.get("id") + method = message.get("method") + + if method == "notifications/initialized": + return None + if method == "initialize": + requested = (message.get("params") or {}).get("protocolVersion") + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "protocolVersion": requested or PROTOCOL_VERSION, + "capabilities": {"tools": {"listChanged": False}}, + "serverInfo": {"name": "mem0", "version": PLUGIN_VERSION}, + }, + } + if method == "ping": + return {"jsonrpc": "2.0", "id": request_id, "result": {}} + if method == "tools/list": + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "tools": [ + { + "name": TOOL_NAME, + "description": TOOL_DESCRIPTION, + "inputSchema": TOOL_SCHEMA, + "annotations": { + "readOnlyHint": True, + "idempotentHint": True, + "openWorldHint": True, + }, + } + ] + }, + } + if method == "tools/call": + params = message.get("params") or {} + if params.get("name") != TOOL_NAME: + result = _tool_response("Unknown Mem0 tool.", is_error=True) + else: + try: + result = _tool_response( + call_search_memories(params.get("arguments"), _workspace_cwd(params)) + ) + except ToolInputError as exc: + result = _tool_response(str(exc), is_error=True) + except Exception: + result = _tool_response("Memory search failed.", is_error=True) + return {"jsonrpc": "2.0", "id": request_id, "result": result} + if request_id is None: + return None + return { + "jsonrpc": "2.0", + "id": request_id, + "error": {"code": -32601, "message": "Method not found"}, + } + + +def main() -> int: + for raw_line in sys.stdin: + try: + message = json.loads(raw_line) + response = handle_request(message) + except json.JSONDecodeError: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32700, "message": "Parse error"}, + } + except Exception: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32603, "message": "Internal error"}, + } + if response is not None: + sys.stdout.write(json.dumps(response, separators=(",", ":")) + "\n") + sys.stdout.flush() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/antigravity-plugin/core/memory_cli.py b/integrations/antigravity-plugin/core/memory_cli.py new file mode 100644 index 000000000..595729feb --- /dev/null +++ b/integrations/antigravity-plugin/core/memory_cli.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Mem0 diagnostics and user controls.""" + +from __future__ import annotations + +import argparse +import json +import os + +import telemetry +from memory_core import ( + EvidenceStore, + api_key, + data_dir, + doctor, + forget_remote_repo, + configure_harness, + resolve_repo, + user_id, +) + + +def _print_status(value: dict) -> None: + last = value.get("last_operation") or {} + print(f"Mem0: {'paused' if value['paused'] else 'active'}") + print(f"Repository: {value['repo_id']}") + print(f"Local data: {value['data_dir']}") + print(f"API key: {'configured' if value['api_key_configured'] else 'missing'}") + print( + "Saved on this computer: " + f"{value['events']} session details, {value['flushes']} memory updates" + ) + print( + f"Used in this repository: {value['retrievals']} memories returned, " + f"{value['sidekick_runs']} sidekick runs" + ) + if last: + item_label = "" + if last["operation"] in {"flush", "flush-retry"}: + item_label = f", {last['item_count']} memories" + operation = ( + "memory update" + if last["operation"] in {"flush", "flush-retry"} + else last["operation"].replace("-", " ") + ) + print( + f"Last {operation}: " + f"{'succeeded' if last['success'] else 'failed'} " + f"({last['duration_ms']:.1f} ms{item_label})" + ) + sidekick = value.get("last_sidekick") or {} + if sidekick: + state = "finished" if sidekick.get("stopped_at") else "started" + print( + "Last sidekick: " + f"{state}, received {sidekick['context_chars']} characters of memory, " + f"agent {sidekick['agent_id']}" + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + subparsers = parser.add_subparsers(dest="command", required=True) + + status = subparsers.add_parser("status") + status.add_argument("--json", action="store_true") + + doctor_parser = subparsers.add_parser("doctor") + doctor_parser.add_argument("--json", action="store_true") + + subparsers.add_parser("pause") + subparsers.add_parser("resume") + + forget = subparsers.add_parser("forget") + forget.add_argument("--remote", action="store_true") + forget.add_argument("--yes", action="store_true") + forget.add_argument("--include-project-memory", action="store_true") + + args = parser.parse_args() + if args.harness: + source_tag = f"{args.harness.replace('-', '_')}_plugin" + configure_harness(args.harness, source_tag=source_tag) + telemetry.init(harness=args.harness, source_tag=source_tag.upper()) + if args.plugin_data_dir: + os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir + store = EvidenceStore() + try: + repo = resolve_repo(os.getcwd()) + telemetry.record("control", repo=repo, action=args.command) + if args.command == "status": + result = { + **store.status(repo.identity), + "repo_id": repo.identity, + "app_id": repo.app_id, + "project_id": repo.project_id, + "directory": repo.directory, + "user_id": user_id(), + "data_dir": str(data_dir()), + "api_key_configured": bool(api_key()), + } + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + _print_status(result) + elif args.command == "doctor": + result = doctor(os.getcwd()) + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + for name, check in result["checks"].items(): + print( + f"{'PASS' if check['ok'] else 'FAIL'} {name}: {check['detail']}" + ) + return 0 if result["ok"] else 1 + elif args.command == "pause": + store.set_setting("paused", "true") + print("Mem0 stopped saving and searching memories.") + elif args.command == "resume": + store.set_setting("paused", "false") + print("Mem0 resumed saving and searching memories.") + elif args.command == "forget": + if not args.yes: + print( + "Refusing to delete data without --yes. Add --remote to also " + "delete this user/repository scope from Mem0." + ) + return 2 + remote_result = ( + forget_remote_repo( + repo, include_project_memory=args.include_project_memory + ) + if args.remote + else None + ) + local_result = store.forget_local_repo(repo.identity) + print( + json.dumps( + {"local": local_result, "remote": remote_result}, + indent=2, + default=str, + ) + ) + if remote_result and remote_result.get("status") == "error": + return 1 + finally: + store.close() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/antigravity-plugin/core/memory_core.py b/integrations/antigravity-plugin/core/memory_core.py new file mode 100644 index 000000000..cf71196b8 --- /dev/null +++ b/integrations/antigravity-plugin/core/memory_core.py @@ -0,0 +1,2645 @@ +#!/usr/bin/env python3 +"""Shared core for Mem0 agent plugins. + +Hooks record small session details locally. When the agent compacts or ends the +session, Mem0 sends the useful parts to the platform so it can create memories. +The agent can search those memories during later work in the repository. +""" + +from __future__ import annotations + +import functools +import hashlib +import json +import math +import os +import re +import sqlite3 +import subprocess +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +import telemetry + +DEFAULT_API_URL = "https://api.mem0.ai" +PLUGIN_VERSION = "0.3.1" + +_harness_name: str = "generic" +_harness_env_prefix: str = "MEM0_PLUGIN" +_harness_data_dir_name: str = "mem0-plugin" +_harness_source_tag: str = "mem0_plugin" + + +def configure_harness( + name: str, + env_prefix: str = "", + data_dir_name: str = "", + source_tag: str = "", +) -> None: + global _harness_name, _harness_env_prefix, _harness_data_dir_name, _harness_source_tag + _harness_name = name + _harness_env_prefix = env_prefix or f"MEM0_{name.upper().replace('-', '_')}" + _harness_data_dir_name = data_dir_name or f"{name}-plugin" + _harness_source_tag = source_tag or f"{name.replace('-', '_')}_plugin" + + +def harness_config() -> dict[str, str]: + return { + "name": _harness_name, + "env_prefix": _harness_env_prefix, + "data_dir_name": _harness_data_dir_name, + "source_tag": _harness_source_tag, + } + + +MAX_COMMAND_CHARS = 2000 +MAX_RESULT_CHARS = 2500 +MAX_EPISODE_CHARS = 12000 +CHECKPOINT_EXCHANGES = 5 +CHECKPOINT_MESSAGES = 10 +CHECKPOINT_SOURCE_CHARS = 40000 +DEFAULT_MAX_CONTEXT_CHARS = 4000 +MAX_EXTRACTION_INPUT_TOKENS = 24000 +MAX_FLUSH_ATTEMPTS = 5 +FORGET_PAGE_SIZE = 100 +FORGET_MAX_PAGES = 50 + +PROJECT_MEMORY_INSTRUCTIONS = """Save concise repository facts that will help anyone with future coding work in this repository. + +A completed change should produce one memory explaining the resulting behavior, where it is implemented when useful, and any important constraints or reasoning. Exploration or accepted decisions may produce separate memories only when they are independently useful. + +A command that failed and was then made to work should produce one memory naming the failing invocation, the error it returned, and the invocation that succeeded. Do not save one-off errors caused by an edit still in progress, transient network failures, or anything a rerun would fix on its own. + +Use the current coding agent's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or the coding agent completed them. Treat subagent responses as supporting repository evidence, not as decisions. + +Write about the repository, not the user, assistant, session, or task. Do not save personal preferences. Do not save a memory that only states which repository, branch, or directory the session worked in. Do not include test results, documentation updates, release notes, or temporary state. + +If nothing useful was established, return no memories.""" + +PERSONAL_MEMORY_INSTRUCTIONS = """Save concise facts about the user that will help in any repository: preferred tools, package managers, languages, coding style, review and communication preferences, and anything the user explicitly asked to be remembered about themselves. + +Write in the third person about the user, not about the repository, the assistant, the session, or the task. Do not save repository facts, project decisions, commands, or what was built. + +Never save that the user has no preferences or that nothing was learned. If nothing was learned about the user, return no memories.""" + +CODING_MEMORY_CATEGORIES = [ + { + "project_knowledge": ( + "What the project is and how its code, APIs, data, files, and " + "components work." + ) + }, + { + "decisions_and_constraints": ( + "Why an approach was chosen, what must remain true, and rules future " + "work must follow." + ) + }, + { + "workflows": ( + "How to run, test, debug, deploy, configure, or otherwise work on the " + "project." + ) + }, + { + "problems_and_fixes": ( + "Bugs, failures, known pitfalls, their causes, and how to fix or avoid " + "them." + ) + }, + { + "results": ( + "Outcomes and measurements from tests, benchmarks, experiments, or " + "investigations." + ) + }, +] +CODING_MEMORY_CATEGORY_NAMES = tuple( + category_name + for category in CODING_MEMORY_CATEGORIES + for category_name in category +) + +TEST_COMMAND_RE = re.compile( + r"(?:^|\s)(?:pytest|py\.test|jest|vitest|go\s+test|cargo\s+test|" + r"npm\s+(?:run\s+)?test|pnpm\s+(?:run\s+)?test|yarn\s+test|" + r"mvn\s+test|gradle\s+test|make\s+test)(?:\s|$)", + re.IGNORECASE, +) +BUILD_COMMAND_RE = re.compile( + r"(?:^|\s)(?:npm|pnpm|yarn)\s+(?:run\s+)?build(?:\s|$)|" + r"(?:^|\s)(?:cargo|go|mvn|gradle|make)\s+build(?:\s|$)", + re.IGNORECASE, +) + +SECRET_PATTERNS = [ + re.compile(r"(?i)(authorization\s*[:=]\s*(?:bearer|token)\s+)[^\s\"']+"), + re.compile( + r"(?i)((?:api[_-]?key|secret[_-]?access[_-]?key|session[_-]?token)\s*[:=]\s*)[^\s\"']+" + ), + re.compile( + r"(?i)((?:access[_-]?token|refresh[_-]?token|password|credential)" + r"\s*[:=]\s*)[^\s&\"']+" + ), + re.compile(r"\b(?:sk|m0|mem0_sk|psk)-[A-Za-z0-9_\-]{12,}\b"), + re.compile(r"\b(?:ASIA|AKIA)[A-Z0-9]{12,}\b"), + re.compile(r"\b(?:ghp_|github_pat_|xox[baprs]-)[A-Za-z0-9_\-]{12,}\b"), + re.compile( + r"-----BEGIN [^-]*PRIVATE KEY-----.*?-----END [^-]*PRIVATE KEY-----", + re.DOTALL, + ), + re.compile( + r'(?i)("(?:api[_-]?key|password|secret(?:[_-]?access[_-]?key)?' + r'|(?:access|refresh|session)[_-]?token|token|authorization|credential' + r')"\s*:\s*")(?:\\.|[^"\\])*' + ), +] + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def redact(value: Any) -> str: + text = ( + value + if isinstance(value, str) + else json.dumps(value, ensure_ascii=False, default=str) + ) + for pattern in SECRET_PATTERNS: + if pattern.groups: + text = pattern.sub(r"\1[REDACTED]", text) + else: + text = pattern.sub("[REDACTED]", text) + return text + + +def bounded(value: Any, limit: int) -> str: + text = redact(value).strip() + if len(text) <= limit: + return text + return text[:limit] + f"\n...[truncated {len(text) - limit} chars]" + + +def _git(cwd: str, *args: str) -> str: + try: + result = subprocess.run( + ["git", "-C", cwd, *args], + check=False, + capture_output=True, + text=True, + timeout=0.5, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return result.stdout.strip() if result.returncode == 0 else "" + + +def _normalize_remote(remote: str) -> str: + remote = remote.strip() + if remote.startswith("git@") and ":" in remote: + host_path = remote[4:].replace(":", "/", 1) + remote = f"https://{host_path}" + if remote.endswith(".git"): + remote = remote[:-4] + if "://" in remote: + parsed = urllib.parse.urlsplit(remote) + hostname = parsed.hostname or "" + if parsed.port: + hostname = f"{hostname}:{parsed.port}" + remote = urllib.parse.urlunsplit( + (parsed.scheme, hostname, parsed.path, parsed.query, parsed.fragment) + ) + return remote.rstrip("/") + + +_WILDCARD_SCOPE = re.compile(r"^\*+$") + + +def _scope_value(raw: str | None) -> str: + """Reject wildcards as identities: they are filter syntax and would widen the scope.""" + value = (raw or "").strip() + return "" if _WILDCARD_SCOPE.match(value) else value + + +SEARCH_SCOPES = ("repo", "dir", "mine") +DEFAULT_SEARCH_SCOPE = "repo" + +def directory_app_id(repo: RepoContext) -> str: + """The app_id of the directory this session runs in: the repository at the root, repository/path below it.""" + return f"{repo.app_id}/{repo.directory}" if repo.directory else repo.app_id + + +def directory_chain(repo: RepoContext) -> list[str]: + """Every directory a memory belongs to, from the top-level folder down to the one it was written in.""" + parts = repo.directory.split("/") if repo.directory else [] + return ["/".join(parts[: index + 1]) for index in range(len(parts))] + + +def _shared_project_ids(repo: RepoContext) -> list[str]: + """Current and pre-upgrade namespaces, shared by recall and explicit deletion.""" + if not repo.identity.startswith("local:") and repo.project_id != repo.app_id: + return [repo.project_id, repo.app_id] + return [repo.project_id] + + +def _search_filters(user: str, repo: RepoContext, scope: str) -> dict[str, Any]: + """Build the scope filter: app_id scopes to the repo, then union shared and personal lanes.""" + app_scope = {"app_id": repo.app_id} + mine = {"AND": [{"user_id": user}, app_scope]} + if scope == "mine": + return mine + projects = [{"AND": [{"agent_id": project_id}, app_scope]} for project_id in _shared_project_ids(repo)] + shared: dict[str, Any] = projects[0] if len(projects) == 1 else {"OR": projects} + if scope == "dir" and repo.directory: + shared = {"AND": [shared, {"metadata": {"dirs": {"contains": repo.directory}}}]} + return {"OR": [shared, mine]} + + +def search_scope() -> str: + configured = ( + _plugin_option("search_scope", "MEM0_CODE_SEARCH_SCOPE") or "" + ).strip().lower() + return configured if configured in SEARCH_SCOPES else DEFAULT_SEARCH_SCOPE + + +def resolve_search_scope(scope: str | None) -> str: + value = (scope or search_scope()).strip().lower() + if value not in SEARCH_SCOPES: + raise ValueError(f"Unknown search scope: {value}") + return value + + +def _legacy_project_map(cwd: str, root: str, raw_remote: str) -> str: + """Return the project name used by the previous Claude Code plugin.""" + try: + data = json.loads((Path.home() / ".mem0" / "project_map.json").read_text()) + except (OSError, json.JSONDecodeError): + return "" + if not isinstance(data, dict): + return "" + + keys = list(dict.fromkeys([cwd, root, os.path.realpath(cwd), os.path.realpath(root)])) + if raw_remote: + keys.append(f"remote:{hashlib.sha256(raw_remote.encode()).hexdigest()[:16]}") + for key in keys: + value = data.get(key) + if isinstance(value, str) and value.strip(): + return value.strip() + return "" + + +def _legacy_project_id(cwd: str, root: str, raw_remote: str, identity: str) -> str: + """Use the repository namespace created by the previous Mem0 plugin.""" + configured = _scope_value(os.environ.get("MEM0_PROJECT_ID")) + if configured: + return configured + + mapped = _scope_value(_legacy_project_map(cwd, root, raw_remote)) + if mapped: + return mapped + + remote = raw_remote or ("" if identity.startswith("local:") else identity) + remote = remote.strip().removesuffix(".git") + for prefix in ("https://", "http://", "ssh://", "git://"): + if remote.startswith(prefix): + remote = remote[len(prefix) :] + break + else: + remote = re.sub(r"^git@", "", remote) + parts = [part for part in remote.replace(":", "/", 1).split("/") if part] + if len(parts) >= 2: + return f"{parts[-2]}-{parts[-1]}".replace("/", "-").replace(":", "-") + if parts: + return parts[-1].replace("/", "-").replace(":", "-") + return os.path.basename(root or cwd) or "unknown" + + +@dataclass(frozen=True) +class RepoContext: + cwd: str + root: str + identity: str + app_id: str + branch: str + head_sha: str + project_id: str = "" + directory: str = "" + + +def _project_id(root: str, identity: str, app_id: str) -> str: + """The shared namespace: includes a host hash so repos with the same owner/name on different hosts stay apart.""" + if not identity.startswith("local:"): + return f"{app_id}-{hashlib.sha256(identity.encode()).hexdigest()[:10]}" + return f"local-{app_id}-{hashlib.sha256(root.encode()).hexdigest()[:10]}" + + +def _relative_directory(cwd: str, root: str) -> str: + relative = os.path.relpath(cwd, root) + return "" if relative == "." or relative.startswith("..") else relative.replace(os.sep, "/") + + +@dataclass(frozen=True) +class MemorySearchResult: + succeeded: bool + matched_count: int + already_shown_count: int + memories: list[dict[str, Any]] + + +@functools.lru_cache(maxsize=64) +def _resolve_repo_cached(cwd: str) -> RepoContext: + given_cwd = cwd + cwd = os.path.realpath(cwd) + given_root = _git(cwd, "rev-parse", "--show-toplevel") or given_cwd + root = os.path.realpath(given_root) + raw_remote = _git(root, "config", "--get", "remote.origin.url") + remote = _normalize_remote(raw_remote) + identity = remote or f"local:{root}" + app_id = _legacy_project_id(given_cwd, given_root, raw_remote, identity) + return RepoContext( + cwd=cwd, + root=root, + identity=identity, + app_id=app_id, + branch=_git(root, "branch", "--show-current") or "detached", + head_sha=_git(root, "rev-parse", "HEAD"), + project_id=_project_id(root, identity, app_id), + directory=_relative_directory(cwd, root), + ) + + +def resolve_repo(cwd: str | None) -> RepoContext: + return _resolve_repo_cached(os.path.abspath(cwd or os.getcwd())) + + +def api_key() -> str: + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return configured + try: + return (data_dir() / "api-key").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def cache_plugin_api_key() -> bool: + """Bridge host's hook-only sensitive config into plugin-owned storage.""" + configured = ( + os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if not configured: + return False + + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + path = directory / "api-key" + temporary = directory / f"api-key.{os.getpid()}.tmp" + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_TRUNC, + 0o600, + ) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + handle.write(configured) + os.replace(temporary, path) + os.chmod(path, 0o600) + finally: + try: + temporary.unlink() + except FileNotFoundError: + pass + return True + + +def clear_stale_api_key_cache() -> bool: + """Drop the cached key file once every configured key source is gone.""" + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return False + path = data_dir() / "api-key" + if not path.exists(): + return False + try: + path.unlink() + except OSError: + return False + return True + + +def detached_process_kwargs(platform: str | None = None) -> dict: + """Keep a spawned worker alive after the coding agent exits, on POSIX and Windows.""" + if (platform or sys.platform) == "win32": + return { + "creationflags": subprocess.DETACHED_PROCESS + | subprocess.CREATE_NEW_PROCESS_GROUP + } + return {"start_new_session": True} + + +def _plugin_option(name: str, fallback: str = "") -> str: + return ( + os.environ.get(f"PLUGIN_OPTION_{name.upper()}") + or os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") + or os.environ.get(fallback) + or "" + ).strip() + + +def user_id() -> str: + return ( + _scope_value(_plugin_option("user_id", "MEM0_CODE_USER_ID")) + or _scope_value(os.environ.get("MEM0_USER_ID")) + or _scope_value(os.environ.get("MEM0_RESOLVED_USER_ID")) + or _scope_value(os.environ.get("USER")) + or _scope_value(os.environ.get("USERNAME")) + or "default" + ) + + +def data_dir() -> Path: + configured = ( + os.environ.get("MEM0_CODE_DATA_DIR") + or os.environ.get("MEM0_PLUGIN_DATA_DIR") + or os.environ.get("PLUGIN_DATA") + or os.environ.get("CLAUDE_PLUGIN_DATA") + ) + return ( + Path(configured).expanduser() if configured else Path.home() / ".mem0" / _harness_data_dir_name + ) + + +def _bool_option(name: str, fallback: str, default: bool = False) -> bool: + value = _plugin_option(name, fallback) + if not value: + return default + return value.lower() in {"1", "true", "yes", "on"} + + +def _int_option(name: str, fallback: str, default: int) -> int: + value = _plugin_option(name, fallback) + try: + return int(value) if value else default + except ValueError: + return default + + + +def _checkpoint_message(event: dict[str, Any]) -> str: + kind = event.get("kind") + payload = event.get("payload") or {} + if kind == "user_prompt": + return redact(payload.get("text", "")).strip() + if kind == "assistant_stop": + transcript_messages = payload.get("transcript_messages") or [] + if isinstance(transcript_messages, list): + text = "\n".join( + str(message.get("content") or "") + for message in transcript_messages + if isinstance(message, dict) and message.get("content") + ) + if text: + return text + return redact(payload.get("text", "")).strip() + if kind == "sidekick_stop": + return redact(payload.get("final_message", "")).strip() + return "" + + +def checkpoint_stats(events: list[dict[str, Any]]) -> tuple[int, int, int]: + """Return completed exchanges, messages, and source characters.""" + completed = sum(event.get("kind") == "assistant_stop" for event in events) + contents = [content for event in events if (content := _checkpoint_message(event))] + return completed, len(contents), sum(len(content) for content in contents) + + +def select_checkpoint_events( + events: list[dict[str, Any]], *, force: bool +) -> list[dict[str, Any]]: + """Select one ordered extraction block without splitting an exchange.""" + for index, event in enumerate(events): + if event.get("kind") != "assistant_stop": + continue + candidate = events[: index + 1] + completed, messages, source_chars = checkpoint_stats(candidate) + if ( + completed >= CHECKPOINT_EXCHANGES + or messages >= CHECKPOINT_MESSAGES + or source_chars >= CHECKPOINT_SOURCE_CHARS + ): + return candidate + return events if force else [] + + +class EvidenceStore: + def __init__(self, path: Path | None = None): + directory = data_dir() if path is None else path.parent + directory.mkdir(parents=True, exist_ok=True) + self.path = path or directory / "evidence.sqlite3" + try: + self._open() + except sqlite3.DatabaseError: + self._quarantine() + self._open() + + def _open(self) -> None: + self.conn = sqlite3.connect(self.path, timeout=10) + self.conn.row_factory = sqlite3.Row + try: + self.conn.execute("PRAGMA journal_mode=WAL") + self.conn.execute("PRAGMA busy_timeout=10000") + self._migrate() + except sqlite3.DatabaseError: + self.conn.close() + raise + + def _quarantine(self) -> None: + """Move an unreadable database aside so capture restarts cleanly.""" + stamp = int(time.time()) + for suffix in ("", "-wal", "-shm"): + source = Path(f"{self.path}{suffix}") + try: + source.replace(f"{self.path}.corrupt-{stamp}{suffix}") + except FileNotFoundError: + continue + except OSError: + try: + source.unlink() + except OSError: + pass + telemetry.record("db_quarantined") + + def close(self) -> None: + self.conn.close() + + def _migrate(self) -> None: + self.conn.executescript( + """ + CREATE TABLE IF NOT EXISTS events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + created_at TEXT NOT NULL, + kind TEXT NOT NULL, + payload_json TEXT NOT NULL, + flush_id TEXT + ); + CREATE INDEX IF NOT EXISTS events_session_idx + ON events(repo_id, session_id, flush_id, id); + + CREATE TABLE IF NOT EXISTS session_scopes ( + session_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + root TEXT NOT NULL, + branch TEXT NOT NULL, + head_sha TEXT NOT NULL, + created_at TEXT NOT NULL, + directory TEXT NOT NULL DEFAULT '' + ); + + CREATE TABLE IF NOT EXISTS flushes ( + packet_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + reason TEXT NOT NULL, + event_start INTEGER NOT NULL, + event_end INTEGER NOT NULL, + status TEXT NOT NULL, + episode_event_id TEXT, + semantic_event_id TEXT, + error TEXT, + attempts INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + CREATE TABLE IF NOT EXISTS retrievals ( + session_id TEXT NOT NULL, + repo_id TEXT NOT NULL, + memory_id TEXT NOT NULL, + injected_at TEXT NOT NULL, + rank INTEGER, + score REAL, + memory_text TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(session_id, repo_id, memory_id) + ); + + CREATE TABLE IF NOT EXISTS operations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + created_at TEXT NOT NULL, + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + operation TEXT NOT NULL, + duration_ms REAL NOT NULL, + success INTEGER NOT NULL, + item_count INTEGER NOT NULL DEFAULT 0, + request_chars INTEGER NOT NULL DEFAULT 0, + response_chars INTEGER NOT NULL DEFAULT 0, + error TEXT + ); + + CREATE TABLE IF NOT EXISTS sidekick_runs ( + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + agent_type TEXT NOT NULL, + started_at TEXT NOT NULL, + stopped_at TEXT, + transcript_path TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + final_message TEXT, + PRIMARY KEY(repo_id, session_id, agent_id) + ); + CREATE INDEX IF NOT EXISTS sidekick_runs_repo_idx + ON sidekick_runs(repo_id, started_at); + + CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + """ + ) + # Remove the pre-0.1.1 no-tools snapshot implementation. The real coding + # sidekick is a native Claude Code agent and stores no state in this DB. + self.conn.executescript( + """ + DROP TABLE IF EXISTS sidekick_calls; + DROP TABLE IF EXISTS sidekick_snapshots; + DROP TABLE IF EXISTS sidekick_state; + DROP TABLE IF EXISTS sidekick_packets; + """ + ) + self._ensure_column("retrievals", "rank", "INTEGER") + self._ensure_column("retrievals", "score", "REAL") + self._ensure_column("retrievals", "memory_text", "TEXT") + self._ensure_column("retrievals", "context_chars", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("flushes", "attempts", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("session_scopes", "directory", "TEXT NOT NULL DEFAULT ''") + self.conn.commit() + + def _ensure_column(self, table: str, column: str, declaration: str) -> None: + columns = { + str(row["name"]) + for row in self.conn.execute(f"PRAGMA table_info({table})").fetchall() + } + if column not in columns: + self.conn.execute(f"ALTER TABLE {table} ADD COLUMN {column} {declaration}") + + def record_event( + self, + repo: RepoContext, + session_id: str, + kind: str, + payload: dict[str, Any], + ) -> int: + cursor = self.conn.execute( + """INSERT INTO events + (repo_id, app_id, session_id, created_at, kind, payload_json) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + repo.app_id, + session_id, + utc_now(), + kind, + json.dumps(payload, ensure_ascii=False, sort_keys=True), + ), + ) + self.conn.commit() + return int(cursor.lastrowid) + + def record_assistant_response(self, repo: RepoContext, session_id: str, message: str) -> None: + """Ignore repeated response hooks until another prompt or a different answer arrives.""" + with self.conn: + # Serialize the check and insert across concurrent Stop and SessionEnd hooks. + self.conn.execute("BEGIN IMMEDIATE") + previous = self.conn.execute( + """SELECT kind, payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind IN ('user_prompt', 'assistant_stop') + ORDER BY id DESC LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if ( + previous is not None + and previous["kind"] == "assistant_stop" + and json.loads(previous["payload_json"]).get("text") == message + ): + return + self.record_event(repo, session_id, "assistant_stop", {"text": message}) + + def repo_for_session(self, session_id: str, cwd: str | None) -> RepoContext: + """Keep one project scope for every hook in a coding-agent session.""" + current = resolve_repo(cwd) + if session_id == "unknown-session": + return current + + with self.conn: + self.conn.execute( + """INSERT OR IGNORE INTO session_scopes + (session_id, repo_id, app_id, root, branch, head_sha, created_at, directory) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + current.identity, + current.app_id, + current.root, + current.branch, + current.head_sha, + utc_now(), + current.directory, + ), + ) + scope = self.conn.execute( + "SELECT * FROM session_scopes WHERE session_id = ?", (session_id,) + ).fetchone() + same_git_repo = ( + current.identity == scope["repo_id"] and bool(current.head_sha) + ) + pinned = current if same_git_repo else resolve_repo(str(scope["root"])) + return RepoContext( + cwd=current.cwd, + root=pinned.root, + identity=str(scope["repo_id"]), + app_id=str(scope["app_id"]), + branch=pinned.branch, + head_sha=pinned.head_sha, + project_id=pinned.project_id, + directory=str(scope["directory"] or ""), + ) + + def prepare_flush( + self, repo: RepoContext, session_id: str, reason: str + ) -> tuple[str, list[dict[str, Any]]] | None: + with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + existing = self.conn.execute( + """SELECT * FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') + ORDER BY created_at LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: + self.conn.execute( + "UPDATE flushes SET status = 'gave-up', updated_at = ? WHERE packet_id = ?", + (utc_now(), existing["packet_id"]), + ) + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status="gave-up", + success=False, + attempts=int(existing["attempts"] or 0), + ) + existing = None + if existing: + if reason != "periodic" and existing["reason"] == "periodic": + self.conn.execute( + "UPDATE flushes SET reason = ?, updated_at = ? WHERE packet_id = ?", + (reason, utc_now(), existing["packet_id"]), + ) + existing_rows = self.conn.execute( + "SELECT * FROM events WHERE flush_id = ? ORDER BY id", + (existing["packet_id"],), + ).fetchall() + if existing_rows: + return str(existing["packet_id"]), [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in existing_rows + ] + + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo.identity, session_id), + ).fetchall() + if not rows: + return None + + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + events = select_checkpoint_events(events, force=reason != "periodic") + if not events: + return None + event_start, event_end = events[0]["id"], events[-1]["id"] + packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" + packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] + now = utc_now() + + self.conn.execute( + """INSERT OR IGNORE INTO flushes + (packet_id, repo_id, app_id, session_id, reason, event_start, + event_end, status, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, 'prepared', ?, ?)""", + ( + packet_id, + repo.identity, + repo.app_id, + session_id, + reason, + event_start, + event_end, + now, + now, + ), + ) + event_ids = [event["id"] for event in events] + placeholders = ", ".join("?" for _ in event_ids) + self.conn.execute( + f"UPDATE events SET flush_id = ? " + f"WHERE id IN ({placeholders}) AND flush_id IS NULL", + (packet_id, *event_ids), + ) + return packet_id, events + + def checkpoint_due(self, repo_id: str, session_id: str) -> bool: + if self.has_inflight_flush(repo_id, session_id): + return False + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo_id, session_id), + ).fetchall() + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + return bool(select_checkpoint_events(events, force=False)) + + def has_inflight_flush(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status IN ('prepared', 'semantic-queued') + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def flush_record(self, packet_id: str) -> dict[str, Any] | None: + row = self.conn.execute( + "SELECT * FROM flushes WHERE packet_id = ?", (packet_id,) + ).fetchone() + return dict(row) if row else None + + def has_unflushed_events(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def unflushed_starts_with_session_start( + self, repo_id: str, session_id: str + ) -> bool: + row = self.conn.execute( + """SELECT kind FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id LIMIT 1""", + (repo_id, session_id), + ).fetchone() + return bool(row and row["kind"] == "session_start") + + def update_flush(self, packet_id: str, **fields: Any) -> None: + allowed = {"status", "episode_event_id", "semantic_event_id", "error"} + updates = {key: value for key, value in fields.items() if key in allowed} + updates["updated_at"] = utc_now() + clause = ", ".join(f"{key} = ?" for key in updates) + failed = str(fields.get("status", "")) in { + "error", + "semantic-failed", + "semantic-timeout", + "semantic-missing", + } + if failed: + clause += ", attempts = attempts + 1" + with self.conn: + self.conn.execute( + f"UPDATE flushes SET {clause} WHERE packet_id = ?", + [*updates.values(), packet_id], + ) + + def unseen( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> list[dict[str, Any]]: + seen = { + row["memory_id"] + for row in self.conn.execute( + "SELECT memory_id FROM retrievals WHERE session_id = ? AND repo_id = ?", + (session_id, repo_id), + ) + } + return [memory for memory in memories if str(memory.get("id", "")) not in seen] + + def mark_injected( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> None: + now = utc_now() + with self.conn: + for rank, memory in enumerate(memories, start=1): + memory_id = str(memory.get("id", "")) + if memory_id: + memory_text = bounded( + memory.get("memory") or memory.get("text") or "", + 4000, + ) + try: + score = float(memory["score"]) + except (KeyError, TypeError, ValueError): + score = None + self.conn.execute( + """INSERT OR IGNORE INTO retrievals + (session_id, repo_id, memory_id, injected_at, rank, + score, memory_text, context_chars) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + repo_id, + memory_id, + now, + rank, + score, + memory_text, + len(memory_text), + ), + ) + + def injected_memories( + self, session_id: str, repo_id: str + ) -> list[dict[str, Any]]: + """Return the exact memories already supplied to the main conversation.""" + rows = self.conn.execute( + """SELECT memory_id, rank, score, memory_text + FROM retrievals + WHERE session_id = ? AND repo_id = ? + ORDER BY COALESCE(rank, 2147483647), injected_at, memory_id""", + (session_id, repo_id), + ).fetchall() + return [ + { + "id": row["memory_id"], + "memory": row["memory_text"], + "score": row["score"], + } + for row in rows + if row["memory_text"] + ] + + def start_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + context_chars: int, + ) -> bool: + """Record one native sidekick instance and whether context was first sent.""" + with self.conn: + cursor = self.conn.execute( + """INSERT OR IGNORE INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + context_chars) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + utc_now(), + context_chars, + ), + ) + return int(cursor.rowcount) > 0 + + def stop_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + transcript_path: str, + final_message: str, + ) -> str: + now = utc_now() + self.conn.execute("BEGIN IMMEDIATE") + try: + if not agent_id: + rows = self.conn.execute( + """SELECT agent_id FROM sidekick_runs + WHERE repo_id = ? AND session_id = ? AND agent_type = ? AND stopped_at IS NULL + LIMIT 2""", + (repo.identity, session_id, agent_type), + ).fetchall() + # Without a host ID, overlapping runs cannot be correlated reliably. + agent_id = rows[0]["agent_id"] if len(rows) == 1 else f"unknown-agent-{time.time_ns()}" + self.conn.execute( + """INSERT INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + stopped_at, transcript_path, final_message) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(repo_id, session_id, agent_id) DO UPDATE SET + stopped_at = excluded.stopped_at, + transcript_path = excluded.transcript_path, + final_message = excluded.final_message""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + now, + now, + bounded(transcript_path, 2000), + redact(final_message).strip(), + ), + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + return agent_id + + def operation( + self, + repo: RepoContext, + session_id: str, + operation: str, + duration_ms: float, + success: bool, + *, + item_count: int = 0, + request_chars: int = 0, + response_chars: int = 0, + error: str = "", + ) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO operations + (created_at, repo_id, session_id, operation, duration_ms, + success, item_count, request_chars, response_chars, error) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + ( + utc_now(), + repo.identity, + session_id, + operation, + duration_ms, + int(success), + item_count, + request_chars, + response_chars, + bounded(error, 1000), + ), + ) + + def has_operation(self, repo_id: str, session_id: str, operation: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM operations + WHERE repo_id = ? AND session_id = ? AND operation = ? + LIMIT 1""", + (repo_id, session_id, operation), + ).fetchone() + return row is not None + + def has_event(self, repo_id: str, session_id: str, kind: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + return row is not None + + def latest_event_payload( + self, repo_id: str, session_id: str, kind: str + ) -> dict[str, Any]: + row = self.conn.execute( + """SELECT payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + ORDER BY id DESC LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + if not row: + return {} + try: + payload = json.loads(row["payload_json"]) + except json.JSONDecodeError: + return {} + return payload if isinstance(payload, dict) else {} + + def setting(self, key: str, default: str = "") -> str: + row = self.conn.execute( + "SELECT value FROM settings WHERE key = ?", (key,) + ).fetchone() + return str(row["value"]) if row else default + + def set_setting(self, key: str, value: str) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO settings(key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at""", + (key, value, utc_now()), + ) + + def is_paused(self) -> bool: + return self.setting("paused", "false").lower() in { + "1", + "true", + "yes", + "on", + } + + def forget_local_repo(self, repo_id: str) -> dict[str, int]: + tables = { + "events": "repo_id", + "session_scopes": "repo_id", + "flushes": "repo_id", + "retrievals": "repo_id", + "operations": "repo_id", + "sidekick_runs": "repo_id", + } + removed: dict[str, int] = {} + with self.conn: + for table, column in tables.items(): + cursor = self.conn.execute( + f"DELETE FROM {table} WHERE {column} = ?", (repo_id,) + ) + removed[table] = max(int(cursor.rowcount), 0) + return removed + + def status(self, repo_id: str) -> dict[str, Any]: + def count(table: str) -> int: + return int( + self.conn.execute( + f"SELECT COUNT(*) FROM {table} WHERE repo_id = ?", (repo_id,) + ).fetchone()[0] + ) + + last_operation = self.conn.execute( + """SELECT created_at, operation, duration_ms, success, item_count, error + FROM operations WHERE repo_id = ? ORDER BY id DESC LIMIT 1""", + (repo_id,), + ).fetchone() + last_sidekick = self.conn.execute( + """SELECT session_id, agent_id, agent_type, started_at, stopped_at, + context_chars + FROM sidekick_runs WHERE repo_id = ? + ORDER BY started_at DESC LIMIT 1""", + (repo_id,), + ).fetchone() + return { + "paused": self.is_paused(), + "events": count("events"), + "flushes": count("flushes"), + "retrievals": count("retrievals"), + "sidekick_runs": count("sidekick_runs"), + "last_operation": dict(last_operation) if last_operation else None, + "last_sidekick": dict(last_sidekick) if last_sidekick else None, + } + + +def _session_id(hook_input: dict[str, Any]) -> str: + return str(hook_input.get("session_id") or "unknown-session") + + +def record_session_start(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + store.record_event( + repo, + session_id, + "session_start", + { + "source": hook_input.get("source", "startup"), + "model": bounded(hook_input.get("model", ""), 200), + "branch": repo.branch, + "head_sha": repo.head_sha, + }, + ) + telemetry.record( + "session_start", + repo=repo, + session_id=session_id, + trigger=bounded(str(hook_input.get("source", "startup")), 60), + model=bounded(hook_input.get("model", ""), 200), + api_key_configured=bool(api_key()), + is_git_repo=not repo.identity.startswith("local:"), + ) + + +def record_user_prompt( + store: EvidenceStore, hook_input: dict[str, Any] +) -> tuple[RepoContext, str, str, bool]: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prompt = redact(hook_input.get("prompt", "")).strip() + is_first_prompt = not store.has_event(repo.identity, session_id, "user_prompt") + store.record_event(repo, session_id, "user_prompt", {"text": prompt}) + return repo, session_id, prompt, is_first_prompt + + +def _tool_result_preview(response: Any) -> str: + if isinstance(response, dict): + selected = {} + for key in ( + "stdout", + "stderr", + "output", + "content", + "error", + "filePath", + "success", + "interrupted", + ): + if key in response: + selected[key] = response[key] + response = selected or {"keys": sorted(response.keys())[:20]} + return bounded(response, MAX_RESULT_CHARS) + + +def tool_payload(hook_input: dict[str, Any], *, failed: bool | None = False) -> dict[str, Any]: + name = str(hook_input.get("tool_name") or "unknown") + tool_input = hook_input.get("tool_input") or {} + if not isinstance(tool_input, dict): + tool_input = {} + payload: dict[str, Any] = { + "tool": name, + "failed": failed, + "duration_ms": hook_input.get("duration_ms"), + "agent_role": "sidekick" if hook_input.get("agent_id") else "main", + } + if hook_input.get("agent_id"): + payload["agent_id"] = bounded(hook_input["agent_id"], 200) + if hook_input.get("agent_type"): + payload["agent_type"] = bounded(hook_input["agent_type"], 200) + + if name in {"Read", "Write", "Edit", "MultiEdit", "NotebookEdit"}: + path = tool_input.get("file_path") or tool_input.get("notebook_path") + if path: + payload["path"] = bounded(path, 1000) + if name in {"Write", "Edit", "MultiEdit", "NotebookEdit"}: + payload["mutation_chars"] = sum( + len(str(tool_input.get(key, ""))) + for key in ("content", "new_string", "new_source", "edits") + ) + elif name == "Bash" or "command" in tool_input: + command = bounded(tool_input.get("command", ""), MAX_COMMAND_CHARS) + payload["command"] = command + payload["command_kind"] = ( + "test" + if TEST_COMMAND_RE.search(command) + else "build" + if BUILD_COMMAND_RE.search(command) + else "shell" + ) + response = ( + hook_input.get("error") if failed else hook_input.get("tool_response") + ) + payload["result_preview"] = _tool_result_preview(response) + elif name in {"Grep", "Glob", "WebSearch", "WebFetch"}: + for key in ("pattern", "path", "query", "url"): + if tool_input.get(key): + payload[key] = bounded(tool_input[key], 1000) + else: + payload["input_keys"] = sorted(tool_input.keys())[:20] + if failed: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + + if failed and "error" not in payload: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + return payload + + +def record_tool( + store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool | None = False +) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + payload = tool_payload(hook_input, failed=failed) + if payload.get("path"): + payload["repo_path"] = _repo_relative_path(repo, str(payload["path"])) + store.record_event( + repo, + session_id, + "tool_failure" if failed else "tool_result", + payload, + ) + + +def record_sidekick_start( + store: EvidenceStore, hook_input: dict[str, Any], *, inject_context: bool = True +) -> str: + """Record a native sidekick and reuse the main turn's retrieved memories.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_id = bounded(hook_input.get("agent_id", "unknown-agent"), 200) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + context = combine_context( + format_context(store.injected_memories(session_id, repo.identity)) + ) + if not inject_context: + context = "" + first_start = store.start_sidekick( + repo, session_id, agent_id, agent_type, len(context) + ) + store.record_event( + repo, + session_id, + "sidekick_start", + { + "agent_id": agent_id, + "agent_type": agent_type, + "context_chars": len(context) if first_start else 0, + "worktree_root": bounded(repo.root, 2000), + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="start", + first_start=first_start, + context_chars=len(context) if first_start else 0, + ) + return context if first_start else "" + + +def record_sidekick_stop(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + agent_id = bounded(hook_input.get("agent_id", ""), 200) + final_message = redact(hook_input.get("last_assistant_message", "")).strip() + transcript_path = bounded(hook_input.get("agent_transcript_path", ""), 2000) + agent_id = store.stop_sidekick( + repo, + session_id, + agent_id, + agent_type, + transcript_path, + final_message, + ) + store.record_event( + repo, + session_id, + "sidekick_stop", + { + "agent_id": agent_id, + "agent_type": agent_type, + "transcript_path": transcript_path, + "final_message": final_message, + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="stop", + has_transcript=bool(transcript_path), + message_chars=len(final_message), + ) + + +def _ordered_unique(values: Iterable[str]) -> list[str]: + seen: set[str] = set() + result = [] + for value in values: + if value and value not in seen: + seen.add(value) + result.append(value) + return result + + +def _repo_relative_path(repo: RepoContext, value: str) -> str: + value = str(value or "").strip() + if not value: + return "" + try: + path = Path(value) + if path.is_absolute(): + try: + return path.resolve().relative_to(Path(repo.root).resolve()).as_posix() + except ValueError: + return "" + except (OSError, ValueError): + pass + return bounded(value, 1000) + + +def _render_command_lines(commands: list[dict[str, str]]) -> list[str]: + lines = [] + for command in commands: + line = f"- [{command['status']}/{command['kind']}] {command['command']}" + if command["result"]: + line += f" — {bounded(command['result'], 500).replace(chr(10), ' ')}" + lines.append(line) + return lines + + +def build_episode( + repo: RepoContext, + session_id: str, + packet_id: str, + events: list[dict[str, Any]], + *, + canonical_task: str = "", + task_outcome: str = "", +) -> tuple[str, dict[str, Any]]: + prompts = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "user_prompt" and e["payload"].get("text") + ] + assistant_conclusions = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "assistant_stop" and e["payload"].get("text") + ] + sidekick_outcomes = [ + redact(e["payload"].get("final_message", "")).strip() + for e in events + if e["kind"] == "sidekick_stop" and e["payload"].get("final_message") + ] + tools = [ + e["payload"] for e in events if e["kind"] in {"tool_result", "tool_failure"} + and e["payload"].get("agent_role", "main") == "main" + ] + read_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") == "Read" + ) + modified_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") in {"Write", "Edit", "MultiEdit", "NotebookEdit"} + ) + searches = [ + {key: t[key] for key in ("tool", "pattern", "path", "query", "url") if key in t} + for t in tools + if t.get("tool") in {"Grep", "Glob", "WebSearch", "WebFetch"} + ] + commands = [ + { + "command": t.get("command", ""), + "kind": t.get("command_kind", "shell"), + "status": "unknown" if t.get("failed", False) is None else "failed" if t.get("failed") else "succeeded", + "result": t.get("result_preview", ""), + } + for t in tools + if t.get("command") + ] + + task = bounded(canonical_task or (prompts[0] if prompts else ""), 4000) + conclusion = redact(assistant_conclusions[-1] if assistant_conclusions else "").strip() + outcome = bounded(task_outcome, 2000) + + extraction_messages: list[dict[str, str]] = [] + pending_user_messages: list[dict[str, str]] = [] + if task and not prompts: + pending_user_messages.append({"role": "user", "content": task}) + for event in events: + if event["kind"] == "user_prompt" and event["payload"].get("text"): + pending_user_messages.append( + { + "role": "user", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + elif event["kind"] == "assistant_stop": + transcript_messages = event["payload"].get("transcript_messages") or [] + if isinstance(transcript_messages, list) and transcript_messages: + transcript_users = { + redact(message.get("content") or "").strip() + for message in transcript_messages + if isinstance(message, dict) and message.get("role") == "user" + } + extraction_messages.extend( + message + for message in pending_user_messages + if message["content"].strip() not in transcript_users + ) + extraction_messages.extend( + { + "role": str(message.get("role") or ""), + "content": redact(message.get("content") or "").strip(), + } + for message in transcript_messages + if isinstance(message, dict) + and message.get("role") in {"user", "assistant"} + and message.get("content") + ) + else: + extraction_messages.extend(pending_user_messages) + if event["payload"].get("text"): + extraction_messages.append( + { + "role": "assistant", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + pending_user_messages = [] + elif event["kind"] == "sidekick_stop": + pass + extraction_messages.extend(pending_user_messages) + + structured = { + "packet_id": packet_id, + "repo": repo.identity, + "app_id": repo.app_id, + "session_id": session_id, + "branch": repo.branch, + "head_sha": repo.head_sha, + "task": task, + "task_outcome": outcome, + "assistant_conclusion": conclusion, + "user_messages": prompts, + "assistant_outcomes": assistant_conclusions, + "sidekick_outcomes": sidekick_outcomes, + "extraction_messages": extraction_messages, + "files_read": read_paths[:50], + "files_modified": modified_paths[:50], + "searches": searches[-30:], + "commands": commands[-30:], + } + lines = ["Coding-session episode"] + if task: + lines.extend(["", "Task:", task]) + if modified_paths: + lines.extend( + ["", "Files modified:", *[f"- {path}" for path in modified_paths[:50]]] + ) + if read_paths: + lines.extend(["", "Files read:", *[f"- {path}" for path in read_paths[:50]]]) + if commands: + lines.append("") + lines.append("Observed commands:") + lines.extend(_render_command_lines(commands[-30:])) + if searches: + lines.extend( + [ + "", + "Observed searches:", + *[ + f"- {json.dumps(item, ensure_ascii=False, sort_keys=True)}" + for item in searches[-20:] + ], + ] + ) + if conclusion: + lines.extend(["", "Agent conclusion:", conclusion]) + if outcome: + lines.extend(["", "Task outcome:", outcome]) + lines.extend( + [ + "", + f"Provenance: repo={repo.identity}; branch={repo.branch}; head={repo.head_sha}; packet={packet_id}", + ] + ) + content = "\n".join(lines) + return bounded(content, MAX_EPISODE_CHARS), structured + + +def build_semantic_evidence(structured: dict[str, Any]) -> str: + """Format changed paths for memory extraction. + + Test and build results remain in the local evidence store for diagnostics, + but are not useful repository knowledge by default and should not steer + memory extraction toward transient verification details. + """ + modified_paths = [ + bounded(path, 500) for path in structured.get("files_modified", [])[:20] + ] + commands = structured.get("commands") or [] + if not any(command.get("status") == "failed" for command in commands): + commands = [] + + if not modified_paths and not commands: + return "" + + lines = ["Additional repository details from this session"] + if modified_paths: + lines.extend( + [ + "", + "Changed paths:", + *[f"- {path}" for path in modified_paths], + ] + ) + if commands: + lines.extend(["", "Commands run in this session:", *_render_command_lines(commands)]) + return bounded("\n".join(lines), 8000) + + +def build_extraction_messages(structured: dict[str, Any]) -> list[dict[str, str]]: + """Build the session messages sent to Mem0 for memory extraction.""" + evidence = build_semantic_evidence(structured) + messages = [ + {"role": message["role"], "content": redact(message["content"]).strip()} + for message in structured.get("extraction_messages", []) + if message.get("role") in {"user", "assistant"} and message.get("content") + ] + if evidence: + for message in reversed(messages): + if message["role"] == "assistant": + message["content"] = f"{message['content']}\n\n{evidence}" + break + else: + messages.append({"role": "assistant", "content": evidence}) + + return messages + + +def _estimated_tokens(value: str) -> int: + """Conservatively estimate tokens without adding a tokenizer dependency.""" + ascii_chars = sum(ord(char) < 128 for char in value) + return math.ceil((ascii_chars * 0.4) + (len(value) - ascii_chars)) + + +def _message_tokens(messages: list[dict[str, str]]) -> int: + return _estimated_tokens(json.dumps(messages, ensure_ascii=False)) + + +def _is_agent_assignment(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent assignment (") + + +def _is_agent_response(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent response (") + + +def extraction_message_batches( + messages: list[dict[str, str]], + *, + max_tokens: int = MAX_EXTRACTION_INPUT_TOKENS, +) -> list[list[dict[str, str]]]: + """Keep exchanges together when possible; split oversized messages to enforce the request budget.""" + if not messages or _message_tokens(messages) <= max_tokens: + return [messages] + + exchanges: list[list[dict[str, str]]] = [] + exchange: list[dict[str, str]] = [] + for message in messages: + if message.get("role") == "user" and exchange: + exchanges.append(exchange) + exchange = [] + exchange.append(message) + if exchange: + exchanges.append(exchange) + + units: list[list[dict[str, str]]] = [] + for exchange in exchanges: + if _message_tokens(exchange) <= max_tokens: + units.append(exchange) + continue + index = 0 + while index < len(exchange): + message = exchange[index] + if ( + _is_agent_assignment(message) + and index + 1 < len(exchange) + and _is_agent_response(exchange[index + 1]) + ): + units.append(exchange[index : index + 2]) + index += 2 + else: + units.append([message]) + index += 1 + + bounded_units: list[list[dict[str, str]]] = [] + for unit in units: + if _message_tokens(unit) <= max_tokens: + bounded_units.append(unit) + continue + for message in unit: + remaining = message["content"] + while remaining: + low, high = 0, len(remaining) + while low < high: + middle = (low + high + 1) // 2 + if _message_tokens([{**message, "content": remaining[:middle]}]) <= max_tokens: + low = middle + else: + high = middle - 1 + if low == 0: + raise ValueError("Extraction token budget cannot fit a message") + bounded_units.append([{**message, "content": remaining[:low]}]) + remaining = remaining[low:] + + batches: list[list[dict[str, str]]] = [] + batch: list[dict[str, str]] = [] + for unit in bounded_units: + candidate = [*batch, *unit] + if batch and _message_tokens(candidate) > max_tokens: + batches.append(batch) + batch = list(unit) + else: + batch = candidate + if batch: + batches.append(batch) + return batches + + +def _request_json( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + raw = json.dumps(payload, ensure_ascii=False).encode() + request = urllib.request.Request( + url, + data=raw, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + parsed = json.loads(response_raw or b"{}") + return parsed, len(raw), len(response_raw) + + +def _request_json_with_network_retry( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + """Retry one transient connection failure without retrying API responses.""" + try: + return _request_json(url, key, payload, timeout) + except urllib.error.HTTPError: + raise + except (urllib.error.URLError, TimeoutError, OSError): + time.sleep(0.25) + return _request_json(url, key, payload, timeout) + + +def _get_json( + url: str, key: str, timeout: float +) -> tuple[dict[str, Any] | list[Any], int]: + request = urllib.request.Request( + url, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="GET", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + return json.loads(response_raw or b"{}"), len(response_raw) + + +def _event_id(response: dict[str, Any] | list[Any]) -> str: + return str(response.get("event_id", "")) if isinstance(response, dict) else "" + + +def _stored_event_ids(value: Any) -> list[str]: + raw = str(value or "") + if not raw.startswith("["): + return [] + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return [] + return [str(item or "") for item in parsed] if isinstance(parsed, list) else [] + + +def _result_count(response: dict[str, Any] | list[Any]) -> int: + if isinstance(response, dict): + results = response.get("results") + return len(results) if isinstance(results, list) else 0 + return len(response) if isinstance(response, list) else 0 + + +def touch_handoff_heartbeat() -> None: + """Mark the worker's handoff file alive so recovery does not relaunch it.""" + path = os.environ.get("MEM0_CODE_HANDOFF_PATH", "") + if not path: + return + try: + os.utime(path) + except OSError: + pass + + +def _wait_for_event(api_url: str, key: str, event_id: str) -> tuple[str, int, int]: + """Wait for extraction to finish before a later task can search the store.""" + if not event_id: + return "MISSING", 0, 0 + wait_seconds = float(os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120")) + poll_seconds = max(float(os.environ.get("MEM0_CODE_EVENT_POLL_SECONDS", "1")), 0.1) + deadline = time.monotonic() + wait_seconds + response_chars = 0 + while time.monotonic() < deadline: + touch_handoff_heartbeat() + try: + response, size = _get_json( + f"{api_url}/v1/event/{event_id}/", + key, + min(10, poll_seconds + 5), + ) + except urllib.error.HTTPError as exc: + if exc.code not in {408, 429} and exc.code < 500: + raise + time.sleep(poll_seconds) + continue + except (urllib.error.URLError, TimeoutError, OSError): + # The extraction job is durable server-side. A transient polling + # failure must not discard a job that may still complete normally. + time.sleep(poll_seconds) + continue + response_chars += size + status = ( + str(response.get("status", "UNKNOWN")) + if isinstance(response, dict) + else "UNKNOWN" + ) + if status in {"SUCCEEDED", "FAILED"}: + return status, response_chars, _result_count(response) + time.sleep(poll_seconds) + return "TIMEOUT", response_chars, 0 + + +def _record_flush( + repo: RepoContext, + session_id: str, + reason: str, + status: str, + elapsed: float, + **extra: Any, +) -> None: + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status=status, + success=status in {"semantic-succeeded", "nothing-to-flush"}, + duration_ms=round(elapsed, 2), + **extra, + ) + + +def flush_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + key = api_key() + if not key: + telemetry.record("flush", reason=reason, status="local-only", success=False) + return {"status": "local-only", "reason": "no-api-key"} + + session_id = _session_id(hook_input) + if session_id == "unknown-session": + telemetry.record("flush", reason=reason, status="no-session-id", success=False) + return {"status": "error", "reason": "no-session-id"} + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prepared = store.prepare_flush(repo, session_id, reason) + if prepared is None: + return {"status": "nothing-to-flush"} + packet_id, events = prepared + existing_flush = store.flush_record(packet_id) or {} + + _, structured = build_episode( + repo, + session_id, + packet_id, + events, + canonical_task=bounded(hook_input.get("task", ""), 4000), + task_outcome=bounded(hook_input.get("task_outcome", ""), 2000), + ) + + metadata = {"source": _harness_source_tag} + if repo.branch and repo.branch not in {"detached", "unknown"}: + metadata["branch"] = repo.branch + if repo.head_sha: + metadata["git_sha"] = repo.head_sha + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + add_url = f"{api_url}/v3/memories/add/" + + write_user = _scope_value(user_id()) + write_project = _scope_value(repo.project_id) + if not write_user or not _scope_value(repo.app_id) or not write_project: + telemetry.record("flush", reason=reason, status="unscoped", success=False) + return {"status": "error", "reason": "wildcard-scope"} + + body = { + "agent_id": write_project, + "user_id": write_user, + "app_id": repo.app_id, + "run_id": session_id, + "metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)}, + "agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS, + "custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS, + "custom_categories": CODING_MEMORY_CATEGORIES, + "infer": True, + } + + started = time.perf_counter() + try: + stored_events = _stored_event_ids(existing_flush.get("semantic_event_id")) + existing_event = ( + "" if stored_events else str(existing_flush.get("semantic_event_id") or "") + ) + if existing_event: + existing_status, existing_resp, existing_items = _wait_for_event( + api_url, key, existing_event + ) + if existing_status == "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=existing_event, + error="", + ) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + True, + item_count=existing_items, + response_chars=existing_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=existing_items, + resumed=True, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "semantic_status": existing_status, + "memory_count": existing_items, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + if existing_status == "TIMEOUT": + elapsed = (time.perf_counter() - started) * 1000 + error = "semantic extraction event timed out" + store.update_flush(packet_id, status="semantic-timeout", error=error) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + False, + response_chars=existing_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-timeout", + elapsed, + resumed=True, + error_kind="timeout", + ) + return { + "status": "semantic-timeout", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + + message_batches = [ + batch + for batch in extraction_message_batches( + build_extraction_messages(structured) + ) + if batch + ] + batches = [(body, messages) for messages in message_batches] + if not batches: + store.update_flush(packet_id, status="semantic-succeeded", error="") + return {"status": "nothing-to-flush", "packet_id": packet_id} + operation_name = "flush-retry" if stored_events else "flush" + semantic_events = stored_events[: len(batches)] + semantic_events += [""] * (len(batches) - len(semantic_events)) + semantic_req = 0 + semantic_resp = 0 + for index, (body, messages) in enumerate(batches): + if semantic_events[index]: + continue + semantic_response, request_chars, response_chars = _request_json( + add_url, + key, + {**body, "messages": messages}, + 15, + ) + semantic_events[index] = _event_id(semantic_response) + semantic_req += request_chars + semantic_resp += response_chars + store.update_flush( + packet_id, + status="semantic-queued", + semantic_event_id=json.dumps(semantic_events), + ) + + semantic_event = semantic_events[-1] + semantic_status = "SUCCEEDED" + event_resp = 0 + semantic_items = 0 + failed_event = semantic_event + for index, queued_event in enumerate(semantic_events): + status, response_chars, item_count = _wait_for_event( + api_url, key, queued_event + ) + event_resp += response_chars + semantic_items += item_count + if status != "SUCCEEDED": + semantic_status = status + failed_event = queued_event + if status in {"FAILED", "MISSING"}: + semantic_events[index] = "" + store.update_flush( + packet_id, semantic_event_id=json.dumps(semantic_events) + ) + break + if semantic_status != "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + error = f"semantic extraction event {semantic_status.lower()}" + store.update_flush( + packet_id, status=f"semantic-{semantic_status.lower()}", error=error + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + False, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + f"semantic-{semantic_status.lower()}", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + error_kind=telemetry.error_kind(error), + ) + return { + "status": f"semantic-{semantic_status.lower()}", + "packet_id": packet_id, + "semantic_event_id": failed_event, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=json.dumps(semantic_events), + error="", + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + True, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + request_chars=semantic_req, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": semantic_event, + "semantic_status": semantic_status, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + except Exception as exc: # hooks must fail open + elapsed = (time.perf_counter() - started) * 1000 + error = bounded(str(exc), 1000) + store.update_flush(packet_id, status="error", error=error) + store.operation(repo, session_id, "flush", elapsed, False, error=error) + _record_flush( + repo, + session_id, + reason, + "error", + elapsed, + error_kind=telemetry.error_kind(exc), + ) + return {"status": "error", "packet_id": packet_id, "error": error} + + +def checkpoint_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + """Run remote extraction at a durable boundary.""" + return flush_session(store, hook_input, reason) + + + +def search_memories( + store: EvidenceStore | None, + repo: RepoContext, + session_id: str | None, + query: str, + *, + top_k: int | None = None, + category: str | None = None, + scope: str | None = None, + run_id: str | None = None, + operation: str = "search", + timeout: float = 5, +) -> MemorySearchResult: + key = api_key() + if not key or not query.strip(): + return MemorySearchResult(False, 0, 0, []) + search_once = os.environ.get( + "MEM0_CODE_SEARCH_ONCE_PER_SESSION", "false" + ).lower() in { + "1", + "true", + "yes", + "on", + } + track_session = store is not None and bool(session_id) + if ( + search_once + and track_session + and store.has_operation(repo.identity, session_id, "search") + ): + return MemorySearchResult(False, 0, 0, []) + + result_limit = min( + max( + top_k + if top_k is not None + else _int_option("top_k", "MEM0_CODE_TOP_K", 3), + 1, + ), + 20, + ) + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ValueError(f"Unknown memory category: {category}") + user, project = _scope_value(user_id()), _scope_value(repo.project_id) + if not user or not project or not _scope_value(repo.app_id): + return MemorySearchResult(False, 0, 0, []) + filters = _search_filters(user, repo, resolve_search_scope(scope)) + if category: + filters = {"AND": [filters, {"categories": {"contains": category}}]} + if run_id: + filters = {"AND": [filters, {"run_id": run_id}]} + payload = { + "query": query, + "app_id": repo.app_id, + "filters": filters, + "top_k": result_limit, + "rerank": False, + "latest_only": True, + } + url = ( + os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + + "/v3/memories/search/" + ) + started = time.perf_counter() + try: + response, request_chars, response_chars = _request_json_with_network_retry( + url, key, payload, timeout + ) + memories = ( + response if isinstance(response, list) else response.get("results", []) + ) + memories = [ + memory + for memory in memories + if isinstance(memory, dict) + and (memory.get("metadata") or {}).get("record_kind") != "task_episode" + ][:result_limit] + if track_session: + returned_memories = store.unseen(session_id, repo.identity, memories) + store.mark_injected(session_id, repo.identity, returned_memories) + already_shown_count = len(memories) - len(returned_memories) + else: + returned_memories = memories + already_shown_count = 0 + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation( + repo, + session_id, + operation, + elapsed, + True, + item_count=len(returned_memories), + request_chars=request_chars, + response_chars=response_chars, + ) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=True, + duration_ms=round(elapsed, 2), + matched_count=len(memories), + returned_count=len(returned_memories), + already_shown_count=already_shown_count, + top_k=result_limit, + has_category=bool(category), + ) + return MemorySearchResult( + succeeded=True, + matched_count=len(memories), + already_shown_count=already_shown_count, + memories=returned_memories, + ) + except Exception as exc: + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation(repo, session_id, operation, elapsed, False, error=str(exc)) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=False, + duration_ms=round(elapsed, 2), + top_k=result_limit, + has_category=bool(category), + error_kind=telemetry.error_kind(exc), + ) + return MemorySearchResult(False, 0, 0, []) + + +def format_context( + memories: list[dict[str, Any]], + heading: str = "Relevant repository memories:", +) -> str: + if not memories: + return "" + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + lines = [heading] if heading else [] + for memory in memories: + text = re.sub( + r"\s+", + " ", + redact(memory.get("memory") or memory.get("text") or ""), + ) + text = text.strip() + if not text: + continue + branch = str((memory.get("metadata") or {}).get("branch") or "").strip() + branch_label = ( + f" [learnt on branch {branch}]" + if branch.casefold() not in {"", "main", "master", "unknown", "detached"} + else "" + ) + number = len(lines) if heading else len(lines) + 1 + entry = f"{number}. {text}{branch_label}" + candidate = "\n".join([*lines, entry]) + if len(candidate) <= limit: + lines.append(entry) + continue + if not lines or (heading and len(lines) == 1): + prefix = f"{number}. " + suffix = f"…{branch_label}" + available = ( + limit + - len("\n".join(lines)) + - (1 if lines else 0) + - len(prefix) + - len(suffix) + ) + if available > 0: + lines.append(prefix + text[:available].rstrip() + suffix) + break + minimum_lines = 2 if heading else 1 + return "\n".join(lines) if len(lines) >= minimum_lines else "" + + +def format_search_result(result: MemorySearchResult) -> str: + """Return only the text the coding agent needs from an explicit memory search.""" + if not result.succeeded: + return "Memory search failed." + if result.memories: + rendered = format_context(result.memories, heading="") + if rendered: + return rendered + return "No matching memories found." + + +def combine_context(*contexts: str) -> str: + """Combine memory sources under one hard budget without repeated lines.""" + seen: set[str] = set() + lines: list[str] = [] + for context in contexts: + for line in str(context or "").splitlines(): + normalized = re.sub(r"\s+", " ", line).strip().casefold() + if not normalized or normalized in seen: + continue + seen.add(normalized) + lines.append(line.rstrip()) + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + return bounded("\n".join(lines), limit) if lines else "" + + +def _scoped_memory_ids( + api_url: str, key: str, user: str, repo: RepoContext, include_project: bool +) -> list[str]: + """List this user's memory ids for this repository, plus the shared project memory when asked.""" + ids: list[str] = [] + seen: set[str] = set() + prefix = repo.app_id + _collect_memory_ids( + api_url, key, {"user_id": user}, ids, seen, + app_id_prefix=prefix, + ) + if include_project: + for project_id in _shared_project_ids(repo): + _collect_memory_ids( + api_url, key, {"agent_id": project_id}, ids, seen, + app_id_prefix=prefix, + ) + return ids + + +def _collect_memory_ids( + api_url: str, + key: str, + filters: dict[str, Any], + ids: list[str], + seen: set[str], + *, + app_id_prefix: str = "", +) -> None: + """Page through one list filter; the list endpoint returns nothing for an OR whose user branch has no memories.""" + payload = {"filters": filters} + for page in range(1, FORGET_MAX_PAGES + 1): + parsed, _, _ = _request_json( + f"{api_url}/v2/memories/?page={page}&page_size={FORGET_PAGE_SIZE}", + key, + payload, + 15, + ) + items = parsed.get("results") if isinstance(parsed, dict) else parsed + if not isinstance(items, list) or not items: + break + for item in items: + if not isinstance(item, dict): + continue + memory_id = str(item.get("id", "")) + if not memory_id or memory_id in seen: + continue + if app_id_prefix: + item_app_id = str(item.get("app_id") or "") + if item_app_id != app_id_prefix and not item_app_id.startswith(app_id_prefix + "/"): + continue + seen.add(memory_id) + ids.append(memory_id) + if len(items) < FORGET_PAGE_SIZE: + break + + +def _delete_memory(api_url: str, key: str, memory_id: str) -> bool: + request = urllib.request.Request( + f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/", + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="DELETE", + ) + try: + with urllib.request.urlopen(request, timeout=15): + return True + except Exception: + return False + + +def forget_remote_repo( + repo: RepoContext, *, include_project_memory: bool = False +) -> dict[str, Any]: + """Delete this user's memories for this repository; project memory is shared, so only on request.""" + key = api_key() + if not key: + telemetry.record("forget", repo=repo, success=False, error_kind="no-api-key") + return {"status": "error", "error": "Mem0 API key is not configured"} + user = _scope_value(user_id()) + if not user or not _scope_value(repo.app_id) or not _scope_value(repo.project_id): + telemetry.record("forget", repo=repo, success=False, error_kind="unscoped") + return { + "status": "error", + "error": "Refusing to forget: the user or repository scope is a wildcard", + } + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + try: + memory_ids = _scoped_memory_ids(api_url, key, user, repo, include_project_memory) + except Exception as exc: + telemetry.record( + "forget", repo=repo, success=False, error_kind=telemetry.error_kind(exc) + ) + return {"status": "error", "error": bounded(str(exc), 1000)} + deleted = sum(_delete_memory(api_url, key, memory_id) for memory_id in memory_ids) + failed = len(memory_ids) - deleted + telemetry.record("forget", repo=repo, success=not failed, item_count=deleted) + if failed: + return { + "status": "partial", + "deleted": deleted, + "failed": failed, + "error": f"{failed} of {len(memory_ids)} memories could not be deleted", + } + return {"status": "deleted", "deleted": deleted} + + +def _doctor_mem0_authentication(repo: RepoContext) -> dict[str, Any]: + """Verify the configured key with one read-only, repository-scoped search.""" + key = api_key() + if not key: + return {"ok": False, "detail": "API key missing"} + payload = { + "query": "Mem0 authentication check", + "filters": { + "AND": [ + {"user_id": user_id()}, + {"app_id": repo.app_id}, + ] + }, + "top_k": 1, + "threshold": 1.0, + "rerank": False, + } + url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + started = time.perf_counter() + try: + _request_json(f"{url}/v3/memories/search/", key, payload, 5) + except Exception as exc: + return {"ok": False, "detail": bounded(str(exc), 300)} + elapsed = (time.perf_counter() - started) * 1000 + return {"ok": True, "detail": f"connected ({elapsed:.0f} ms)"} + + +def _doctor_user_id() -> dict[str, Any]: + """Flag a configured user ID the plugin refuses, since the silent fallback surprises people.""" + configured = _plugin_option("user_id", "MEM0_CODE_USER_ID") or os.environ.get( + "MEM0_USER_ID", "" + ) + if configured and not _scope_value(configured): + return { + "ok": False, + "detail": f"configured user_id {configured!r} is a wildcard; using {user_id()!r}", + } + return {"ok": True, "detail": user_id()} + + +def doctor(cwd: str | None = None) -> dict[str, Any]: + repo = resolve_repo(cwd) + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + checks: dict[str, dict[str, Any]] = { + "python": { + "ok": tuple(sys.version_info[:2]) >= (3, 10), + "detail": f"{sys.version_info.major}.{sys.version_info.minor}", + }, + "data_directory": { + "ok": os.access(directory, os.W_OK), + "detail": str(directory), + }, + "mem0_api_key": { + "ok": bool(api_key()), + "detail": "configured" if api_key() else "missing", + }, + "repository": { + "ok": bool(repo.identity), + "detail": repo.identity, + }, + "user_id": _doctor_user_id(), + "mem0_authentication": _doctor_mem0_authentication(repo), + } + return { + "ok": all(bool(value["ok"]) for value in checks.values()), + "plugin_version": PLUGIN_VERSION, + "repo_id": repo.identity, + "app_id": repo.app_id, + "user_id": user_id(), + "checks": checks, + } diff --git a/integrations/antigravity-plugin/core/telemetry.py b/integrations/antigravity-plugin/core/telemetry.py new file mode 100644 index 000000000..249595475 --- /dev/null +++ b/integrations/antigravity-plugin/core/telemetry.py @@ -0,0 +1,397 @@ +#!/usr/bin/env python3 +"""Anonymous usage telemetry for Mem0 agent plugins. + +Hooks run on a 3-6 second budget and fire on every tool call, so recording never +touches the network: `record` appends one JSON line to a local spool and returns. +A detached `python3 telemetry.py` drains the spool in one batched PostHog request, +started once per session and again from the flush worker that is already detached. + +Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false. + +Never sends prompts, memory text, queries, file paths, repository names, or API +keys: only event names, durations, counts, coarse outcomes, and salted hashes. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import platform +import subprocess +import sys +import time +import urllib.error +import urllib.request +import uuid +from pathlib import Path +from typing import Any + +import memory_core + +_harness: str = "generic" +_source_tag: str = "MEM0_PLUGIN" +_PRIVATE_KEYS = { + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +} + + +def init(harness: str = "generic", source_tag: str = "") -> None: + global _harness, _source_tag + _harness = harness + _source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN" + +POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" +POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" +POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/" +EVENT_PREFIX = "code" +SPOOL_LIMIT_BYTES = 256 * 1024 +BATCH_SIZE = 100 +SEND_TIMEOUT = 5 +CLAIM_STALE_SECONDS = 120 +CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60 + + +def is_enabled() -> bool: + """Whether telemetry is switched on for this process.""" + return os.environ.get("MEM0_TELEMETRY", "true").strip().lower() not in { + "false", + "0", + "no", + "off", + } + + +def _digest(value: str, length: int = 16) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length] + + +def _safe_value(value: Any) -> Any: + if isinstance(value, str): + return memory_core.redact(value) + if isinstance(value, dict): + return { + key: _safe_value(item) + for key, item in value.items() + if "".join(character for character in str(key).lower() if character.isalnum()) + not in _PRIVATE_KEYS + } + if isinstance(value, (list, tuple)): + return [_safe_value(item) for item in value] + if value is None or isinstance(value, (bool, int, float)): + return value + return memory_core.redact(value) + + +def _spool_path() -> Path: + return memory_core.data_dir() / "telemetry.jsonl" + + +def _identity_path() -> Path: + return memory_core.data_dir() / "telemetry-identity.json" + + +def _read_identity() -> dict[str, str]: + try: + value = json.loads(_identity_path().read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + return value if isinstance(value, dict) else {} + + +def _write_identity(identity: dict[str, str]) -> None: + path = _identity_path() + temporary = path.with_suffix(f".{os.getpid()}.tmp") + try: + path.parent.mkdir(parents=True, exist_ok=True) + temporary.write_text(json.dumps(identity), encoding="utf-8") + temporary.replace(path) + except OSError: + try: + temporary.unlink() + except OSError: + pass + + +def anonymous_id(identity: dict[str, str] | None = None) -> str: + """Per-machine anonymous identifier, created and persisted on first use.""" + identity = _read_identity() if identity is None else identity + existing = identity.get("anonymous_id") + if existing: + return existing + created = f"code-anon-{uuid.uuid4().hex}" + identity["anonymous_id"] = created + _write_identity(identity) + return created + + +def is_first_run() -> bool: + """Whether this machine has never recorded a plugin event before.""" + return not _identity_path().exists() + + +def record( + event: str, + *, + repo: Any = None, + session_id: str | None = None, + **properties: Any, +) -> None: + """Append one event to the local spool. Never blocks and never raises.""" + if not is_enabled(): + return + try: + spool = _spool_path() + try: + if spool.stat().st_size > SPOOL_LIMIT_BYTES: + return + except OSError: + pass + properties = _safe_value(properties) + properties.update( + harness=_harness, + plugin_version=memory_core.PLUGIN_VERSION, + os=sys.platform, + python_version=platform.python_version(), + ) + if repo is not None: + properties["repo_hash"] = _digest(getattr(repo, "identity", "")) + if session_id: + properties["session_hash"] = _digest(session_id) + line = json.dumps( + { + "event": f"{EVENT_PREFIX}.{event}", + "timestamp": memory_core.utc_now(), + "properties": { + key: value for key, value in properties.items() if value is not None + }, + }, + separators=(",", ":"), + default=str, + ) + spool.parent.mkdir(parents=True, exist_ok=True) + with spool.open("a", encoding="utf-8") as handle: + handle.write(line + "\n") + except Exception: + pass + + +def error_kind(exc: BaseException | str) -> str: + """Coarse, content-free label for a failure, safe to send.""" + text = exc if isinstance(exc, str) else f"{type(exc).__name__}: {exc}" + lowered = text.lower() + if "timed out" in lowered or "timeout" in lowered: + return "timeout" + if "401" in lowered or "403" in lowered or "unauthor" in lowered or "forbidden" in lowered: + return "auth" + if "429" in lowered or "rate limit" in lowered: + return "rate-limited" + if any(code in lowered for code in ("500", "502", "503", "504")): + return "server-error" + if "400" in lowered or "422" in lowered: + return "bad-request" + if isinstance(exc, str): + return "other" + if isinstance(exc, urllib.error.URLError): + return "network" + return type(exc).__name__ + + +def spawn_flush() -> bool: + """Start the detached sender that drains the spool.""" + if not is_enabled(): + return False + try: + if not _spool_path().exists() and not any( + memory_core.data_dir().glob("telemetry-*.sending") + ): + return False + subprocess.Popen( + [sys.executable, str(Path(__file__).resolve())], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + close_fds=True, + **memory_core.detached_process_kwargs(), + ) + return True + except Exception: + return False + + +def _claim_spool() -> Path | None: + """Rename the spool aside so exactly one sender owns each batch.""" + directory = memory_core.data_dir() + claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending" + spool = _spool_path() + try: + spool.replace(claim) + return claim + except OSError: + pass + now = time.time() + for orphan in sorted(directory.glob("telemetry-*.sending")): + try: + age = now - orphan.stat().st_mtime + except OSError: + continue + if age > CLAIM_EXPIRY_SECONDS: + try: + orphan.unlink() + except OSError: + pass + continue + if age < CLAIM_STALE_SECONDS: + continue + try: + orphan.replace(claim) + return claim + except OSError: + continue + return None + + +def _resolve_email(key: str) -> str: + """Trade the API key for the account email so events join other Mem0 surfaces.""" + url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/" + request = urllib.request.Request( + url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json"} + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT) as response: + payload = json.loads(response.read().decode("utf-8")) + except Exception: + return "" + email = payload.get("user_email") if isinstance(payload, dict) else "" + return email if isinstance(email, str) else "" + + +def _post(payload: dict[str, Any], url: str) -> bool: + request = urllib.request.Request( + url, + data=json.dumps(payload, default=str).encode("utf-8"), + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT): + return True + except Exception: + return False + + +def resolve_distinct_id() -> tuple[str, str]: + """Return the PostHog distinct id and the anonymous id it replaced, if any.""" + identity = _read_identity() + email = identity.get("email", "") + if email: + return email, "" + key = memory_core.api_key() + if not key: + return anonymous_id(identity), "" + email = _resolve_email(key) + if not email: + return anonymous_id(identity), "" + previous = identity.get("anonymous_id", "") + identity["email"] = email + _write_identity(identity) + return email, previous + + +def flush() -> int: + """Drain claimed spools to PostHog and return the number of events sent.""" + if not is_enabled(): + return 0 + claim = _claim_spool() + if claim is None: + return 0 + try: + lines = claim.read_text(encoding="utf-8").splitlines() + except OSError: + return 0 + events = [] + for line in lines: + try: + value = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(value, dict) and value.get("event"): + events.append(value) + if not events: + try: + claim.unlink() + except OSError: + pass + return 0 + + distinct_id, aliased_anonymous_id = resolve_distinct_id() + if aliased_anonymous_id: + _post( + { + "api_key": POSTHOG_API_KEY, + "event": "$identify", + "distinct_id": distinct_id, + "properties": { + "$anon_distinct_id": aliased_anonymous_id, + "$lib": "posthog-python", + }, + }, + POSTHOG_CAPTURE_URL, + ) + + sent = 0 + for start in range(0, len(events), BATCH_SIZE): + batch = [ + { + "event": event["event"], + "distinct_id": distinct_id, + "timestamp": event.get("timestamp"), + "properties": { + "source": _source_tag, + "language": "python", + "$process_person_profile": False, + "$lib": "posthog-python", + **(event.get("properties") or {}), + }, + } + for event in events[start : start + BATCH_SIZE] + ] + if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL): + return sent + sent += len(batch) + try: + claim.unlink() + except OSError: + pass + return sent + + +def main() -> int: + flush() + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception: + raise SystemExit(0) diff --git a/integrations/antigravity-plugin/hooks.json b/integrations/antigravity-plugin/hooks.json new file mode 100644 index 000000000..bc2b06f5f --- /dev/null +++ b/integrations/antigravity-plugin/hooks.json @@ -0,0 +1,18 @@ +{ + "mem0": { + "PreInvocation": [ + { "type": "command", "command": "python3 ./hooks/adapter.py PreInvocation", "timeout": 5 } + ], + "PostToolUse": [ + { + "matcher": "*", + "hooks": [ + { "type": "command", "command": "python3 ./hooks/adapter.py PostToolUse", "timeout": 3 } + ] + } + ], + "Stop": [ + { "type": "command", "command": "python3 ./hooks/adapter.py Stop", "timeout": 5 } + ] + } +} diff --git a/integrations/antigravity-plugin/hooks/adapter.py b/integrations/antigravity-plugin/hooks/adapter.py new file mode 100644 index 000000000..8c20e269b --- /dev/null +++ b/integrations/antigravity-plugin/hooks/adapter.py @@ -0,0 +1,163 @@ +#!/usr/bin/env python3 +"""Translate Antigravity hooks into the shared Mem0 runtime.""" + +from __future__ import annotations + +import contextlib +import io +import json +import os +import re +import sys +from pathlib import Path + +HERE = Path(__file__).resolve() +BUNDLED_CORE = HERE.parent.parent / "core" +CORE = BUNDLED_CORE if BUNDLED_CORE.is_dir() else HERE.parents[2] / "core" / "python" +sys.path.insert(0, str(CORE)) + +import hook_runner # noqa: E402 +import telemetry # noqa: E402 +from memory_core import ( # noqa: E402 + configure_harness, + record_tool, + redact, +) + + +def _read_transcript(path: str) -> list[dict[str, str]]: + messages = [] + try: + with open(path, encoding="utf-8") as transcript: + for line in transcript: + try: + step = json.loads(line) + except json.JSONDecodeError: + continue + content = step.get("content") + if not isinstance(content, str) or step.get("status") != "DONE": + continue + if step.get("type") == "USER_INPUT": + match = re.search(r"\s*(.*?)\s*", content, re.DOTALL) + messages.append({"role": "user", "content": redact(match.group(1) if match else content).strip()}) + elif step.get("source") == "MODEL" and step.get("type") == "PLANNER_RESPONSE": + messages.append({"role": "assistant", "content": redact(content).strip()}) + except (OSError, TypeError): + pass + return messages + + +def _transcript_messages(path: str) -> tuple[str, str]: + messages = _read_transcript(path) + return tuple(next((m["content"] for m in reversed(messages) if m["role"] == role), "") + for role in ("user", "assistant")) + + +def _record_stop(store, payload): + session_id = str(payload.get("session_id") or "unknown-session") + repo = store.repo_for_session(session_id, payload.get("cwd")) + path = str(payload.get("transcript_path") or "") + messages = _read_transcript(path) + if not messages: + return hook_runner.default_record_stop(store, payload) + with store.conn: + store.conn.execute("BEGIN IMMEDIATE") + previous = store.latest_event_payload(repo.identity, session_id, "assistant_stop") + offset = previous.get("transcript_count", 0) if previous.get("transcript_path") == path else 0 + if not isinstance(offset, int) or not 0 <= offset <= len(messages): + offset = 0 + if messages[offset:]: + store.record_event(repo, session_id, "assistant_stop", { + "transcript_messages": messages[offset:], + "transcript_count": len(messages), + "transcript_path": path, + }) + return repo, session_id + + +def normalize(payload: dict) -> dict: + value = dict(payload) + value.setdefault("session_id", value.get("conversationId", "")) + workspaces = value.get("workspacePaths") or [] + cwd = workspaces[0] if workspaces else os.environ.get("MEM0_CWD", "").strip() + if cwd: + value.setdefault("cwd", cwd) + value.setdefault("transcript_path", value.get("transcriptPath", "")) + prompt, assistant = _transcript_messages(value["transcript_path"]) + if prompt: + value.setdefault("prompt", prompt) + if assistant: + value.setdefault("last_assistant_message", assistant) + tool_call = value.get("toolCall") or {} + if isinstance(tool_call, dict): + value.setdefault("tool_name", tool_call.get("name", "")) + value.setdefault("tool_input", tool_call.get("args", {})) + if value.get("error"): + value.setdefault("tool_response", value["error"]) + return value + + +def _record_failure(store, payload): + return record_tool(store, payload, failed=True) + + +def _run_shared(arguments: list[str], payload: dict) -> tuple[int, str]: + sys.argv = [sys.argv[0], *arguments] + sys.stdin = io.StringIO(json.dumps(payload)) + output = io.StringIO() + with contextlib.redirect_stdout(output): + result = hook_runner.run( + record_stop_fn=_record_stop, + extra_actions={"post-tool-failure": _record_failure}, + automatic_flush_reasons={"session-end"}, + ) + return result, output.getvalue() + + +def main() -> int: + if len(sys.argv) != 2 or sys.argv[1] not in {"PreInvocation", "PostToolUse", "Stop"}: + return 2 + event = sys.argv[1] + try: + raw = json.load(sys.stdin) + except (json.JSONDecodeError, OSError): + raw = {} + payload = normalize(raw if isinstance(raw, dict) else {}) + if not payload.get("cwd"): + output = {"injectSteps": []} if event == "PreInvocation" else {} + if event == "Stop": + output = {"decision": "allow"} + print(json.dumps(output)) + return 0 + configure_harness("antigravity", data_dir_name="antigravity-plugin", source_tag="antigravity_plugin") + telemetry.init(harness="antigravity", source_tag="ANTIGRAVITY_PLUGIN") + if event == "PreInvocation": + if payload.get("invocationNum") != 0: + print(json.dumps({"injectSteps": []})) + return 0 + _run_shared(["session-start"], payload) + result, output = _run_shared(["user-prompt"], payload) + context = "" + for line in output.splitlines(): + try: + context = json.loads(line)["hookSpecificOutput"]["additionalContext"] + except (json.JSONDecodeError, KeyError, TypeError): + continue + print(json.dumps({"injectSteps": [{"ephemeralMessage": context}] if context else []})) + return result + action = { + "PostToolUse": ["post-tool-failure" if payload.get("error") else "post-tool"], + "Stop": ["flush", "--reason", "session-end"], + }[event] + result, _ = _run_shared(action, payload) + print(json.dumps({"decision": "allow"} if event == "Stop" else {})) + return result + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception as exc: + hook_runner.log_failure(exc) + print("{}") + raise SystemExit(0) from None diff --git a/integrations/antigravity-plugin/mcp_config.json b/integrations/antigravity-plugin/mcp_config.json new file mode 100644 index 000000000..cc9f44de1 --- /dev/null +++ b/integrations/antigravity-plugin/mcp_config.json @@ -0,0 +1,8 @@ +{ + "mcpServers": { + "mem0": { + "command": "python3", + "args": ["./core/mcp_server.py"] + } + } +} diff --git a/integrations/antigravity-plugin/plugin-build.json b/integrations/antigravity-plugin/plugin-build.json new file mode 100644 index 000000000..506bfeb51 --- /dev/null +++ b/integrations/antigravity-plugin/plugin-build.json @@ -0,0 +1,15 @@ +{ + "id": "mem0", + "version": "0.3.1", + "homepage": "https://docs.mem0.ai/integrations/antigravity", + "native": { + "pluginRoot": "${ANTIGRAVITY_PLUGIN_ROOT}", + "files": { + "plugin.json": "plugin.json", + "hooks.json": "hooks.json", + "mcp_config.json": "mcp_config.json", + "hooks/adapter.py": "hooks/adapter.py", + "agents/sidekick/agent.md": "agents/sidekick/agent.md" + } + } +} diff --git a/integrations/antigravity-plugin/plugin.json b/integrations/antigravity-plugin/plugin.json new file mode 100644 index 000000000..32490cc76 --- /dev/null +++ b/integrations/antigravity-plugin/plugin.json @@ -0,0 +1,5 @@ +{ + "$schema": "https://antigravity.google/schemas/v1/plugin.json", + "name": "mem0", + "description": "Cross-session memory and token savings for coding agents." +} diff --git a/integrations/antigravity-plugin/skills/forget/SKILL.md b/integrations/antigravity-plugin/skills/forget/SKILL.md new file mode 100644 index 000000000..a0c4cbe5c --- /dev/null +++ b/integrations/antigravity-plugin/skills/forget/SKILL.md @@ -0,0 +1,26 @@ +--- +name: forget +description: Delete the Mem0 memories stored for this repository and this user. Use when the user asks to forget, clear, wipe, or delete memories. +disable-model-invocation: true +--- + +# Forget this repository's memories + +This permanently deletes remote memories. Before running anything, tell the +user exactly what will be deleted: their own memories for this repository +only. The repository's project memory is shared by everyone who works in it, +so it stays unless the user explicitly asks to delete that too. + +After the user confirms, run: + +```bash +python3 "${ANTIGRAVITY_PLUGIN_ROOT}/core/memory_cli.py" --harness "antigravity" forget --remote --yes +``` + +If the user also asked to delete the repository's shared project memory, add +`--include-project-memory` and say that this removes it for every teammate. + +Report what the command output says was deleted. If the user only wants local +data cleared (evidence log, pending queue), run the same command without +`--remote`. Never pass `--yes` before the user has confirmed in this +conversation. diff --git a/integrations/antigravity-plugin/skills/pause/SKILL.md b/integrations/antigravity-plugin/skills/pause/SKILL.md new file mode 100644 index 000000000..e88f7d315 --- /dev/null +++ b/integrations/antigravity-plugin/skills/pause/SKILL.md @@ -0,0 +1,20 @@ +--- +name: pause +description: Pause Mem0 memory capture on this machine. Use when the user wants to stop memories being recorded, for example for private work or experiments. +disable-model-invocation: true +--- + +# Pause memory capture + +To pause (hooks stop capturing and sending session content; a minimal +anonymous telemetry ping still fires at session start unless +`MEM0_TELEMETRY=false`): + +```bash +python3 "${ANTIGRAVITY_PLUGIN_ROOT}/core/memory_cli.py" --harness "antigravity" pause +``` + +Confirm the new state back to the user, and remind them that already-created +memories still exist and remain searchable. Pending unsent packets are held +while paused, not expired, and are delivered after resuming. To turn capture +back on, use `/mem0:resume`. diff --git a/integrations/antigravity-plugin/skills/remember/SKILL.md b/integrations/antigravity-plugin/skills/remember/SKILL.md new file mode 100644 index 000000000..7be5fc9be --- /dev/null +++ b/integrations/antigravity-plugin/skills/remember/SKILL.md @@ -0,0 +1,21 @@ +--- +name: remember +description: Acknowledge a "remember this" request and make sure it is captured well. Use when the user explicitly asks to remember, note, or save something for future sessions. +disable-model-invocation: true +--- + +# Remember something for future sessions + +Mem0 creates memories from the session automatically — there is no separate +write command. When the user asks to remember something: + +1. Restate the fact clearly and completely in your reply, in one or two + sentences, including any names, values, or paths it depends on. Your visible + reply is what memory extraction reads, so a precise restatement is what gets + remembered. +2. Tell the user it will be saved with this session's memories when the session + ends or compacts, and that it will surface in future sessions in this + repository (they can check later with /mem0:search). + +Do not invent a storage confirmation or a memory ID — creation happens in the +background after the session. diff --git a/integrations/antigravity-plugin/skills/resume/SKILL.md b/integrations/antigravity-plugin/skills/resume/SKILL.md new file mode 100644 index 000000000..52a88a38d --- /dev/null +++ b/integrations/antigravity-plugin/skills/resume/SKILL.md @@ -0,0 +1,19 @@ +--- +name: resume +description: Resume Mem0 memory capture after it was paused with /mem0:pause. +disable-model-invocation: true +--- + +# Resume memory capture + +Resume memory capture for this machine. + +Run: + +```bash +python3 "${ANTIGRAVITY_PLUGIN_ROOT}/core/memory_cli.py" --harness "antigravity" resume +``` + +Confirm to the user that capture is active again. New sessions record evidence and +create memories as normal; nothing that happened while paused is retroactively +captured. diff --git a/integrations/antigravity-plugin/skills/search/SKILL.md b/integrations/antigravity-plugin/skills/search/SKILL.md new file mode 100644 index 000000000..7a6af05fa --- /dev/null +++ b/integrations/antigravity-plugin/skills/search/SKILL.md @@ -0,0 +1,28 @@ +--- +name: search +description: Search memories from earlier Antigravity sessions in this repository. Use it when earlier work may already explain the code, error, decision, or command you need, so you can avoid repeating file reads, searches, or experiments. +argument-hint: "[question] [--top-k number] [--category category-name] [--scope repo|dir|mine] [--run-id session-id]" +disable-model-invocation: true +--- + +# Search memories + +Call `search_memories` with the user's question. Treat `--top-k`, `--category`, +`--scope`, and `--run-id` as tool arguments instead of including them in the +query. + +Omit `top_k` to use Mem0's configured default. Omit `category` to search every +category; a category is a best-effort label Mem0 assigned when it saved the +memory, so if a category search misses, repeat it without the category. Omit +`scope` to use the configured default, normally `repo`: this repository's +shared memory, which everyone who works in it contributes to, plus your own +preferences. + +Pass `scope` when the question needs something else: `dir` to narrow the +shared memory to the directory you are working in (a package inside a +monorepo), `mine` for your own preferences alone. + +Pass `run_id` with any scope to retrieve memories saved in a specific coding-agent +session. Omit `run_id` to search across sessions. It filters the memories returned; +it does not identify the session making the search request. Use a known session ID, +never invent one. Return the tool's result directly. diff --git a/integrations/antigravity-plugin/skills/status/SKILL.md b/integrations/antigravity-plugin/skills/status/SKILL.md new file mode 100644 index 000000000..d417e4a40 --- /dev/null +++ b/integrations/antigravity-plugin/skills/status/SKILL.md @@ -0,0 +1,23 @@ +--- +name: status +description: Show whether Mem0 memory is working in this repository, covering configuration, capture state, pending flushes, and whether the Mem0 API key is valid. Use when the user asks whether memory is on, why a memory is missing, or anything looks broken. +disable-model-invocation: false +--- + +# Memory status + +Run both commands and report the combined result in plain language: + +```bash +python3 "${ANTIGRAVITY_PLUGIN_ROOT}/core/memory_cli.py" --harness "antigravity" status --json +python3 "${ANTIGRAVITY_PLUGIN_ROOT}/core/memory_cli.py" --harness "antigravity" doctor +``` + +Summarize, using only fields the JSON actually reports: whether capture is +active or paused, the user ID and repository scope (`repo_id`), whether an +API key is configured, the event/flush/retrieval counts (`flushes` is the +number of completed flushes, not a pending count), and the doctor check +results. If doctor reports an authentication failure (401 / invalid key), say +clearly that the Mem0 API key is invalid or expired and that memories are NOT +being created. Never report an auth failure as "no memories found". Suggest +reinstalling with `--config api_key=...` in that case. diff --git a/integrations/antigravity-plugin/tests/test_antigravity.py b/integrations/antigravity-plugin/tests/test_antigravity.py new file mode 100644 index 000000000..531a7b181 --- /dev/null +++ b/integrations/antigravity-plugin/tests/test_antigravity.py @@ -0,0 +1,140 @@ +from __future__ import annotations + +import importlib.util +import io +import json +import sys +from pathlib import Path + +HOST = Path(__file__).resolve().parents[1] +CORE_ROOT = HOST.parent / "agent-plugin-core" +sys.path.insert(0, str(CORE_ROOT)) + +from build.build import build # noqa: E402 + +SPEC = importlib.util.spec_from_file_location("antigravity_adapter", HOST / "hooks" / "adapter.py") +assert SPEC and SPEC.loader +adapter = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(adapter) + + +def test_normalizes_antigravity_camel_case_payload(tmp_path: Path) -> None: + transcript = tmp_path / "transcript.jsonl" + transcript.write_text( + "\n".join( + [ + json.dumps( + { + "source": "USER_EXPLICIT", + "type": "USER_INPUT", + "status": "DONE", + "content": "\nremember the parser\n", + } + ), + json.dumps( + { + "source": "MODEL", + "type": "PLANNER_RESPONSE", + "status": "DONE", + "content": "The parser is fixed.", + } + ), + ] + ), + encoding="utf-8", + ) + value = adapter.normalize( + { + "conversationId": "conversation-1", + "workspacePaths": ["/repo"], + "transcriptPath": str(transcript), + "toolCall": {"name": "run_command", "args": {"CommandLine": "pytest"}}, + "error": "failed", + } + ) + + assert value["session_id"] == "conversation-1" + assert value["cwd"] == "/repo" + assert value["transcript_path"] == str(transcript) + assert value["prompt"] == "remember the parser" + assert value["last_assistant_message"] == "The parser is fixed." + assert value["tool_name"] == "run_command" + assert value["tool_input"] == {"CommandLine": "pytest"} + assert value["tool_response"] == "failed" + + +def test_uses_explicit_cwd_when_antigravity_omits_workspaces(monkeypatch) -> None: + monkeypatch.setenv("MEM0_CWD", "/repo") + + value = adapter.normalize({"workspacePaths": []}) + + assert value["cwd"] == "/repo" + + +def test_skips_capture_when_workspace_is_unknown(monkeypatch, capsys) -> None: + monkeypatch.delenv("MEM0_CWD", raising=False) + + def run_shared(*_): + raise AssertionError("shared runtime should not run") + + monkeypatch.setattr(adapter, "_run_shared", run_shared) + monkeypatch.setattr(sys, "argv", ["adapter.py", "PostToolUse"]) + monkeypatch.setattr(sys, "stdin", io.StringIO('{"workspacePaths": []}')) + + assert adapter.main() == 0 + assert json.loads(capsys.readouterr().out) == {} + + +def test_pre_invocation_translates_shared_recall_to_ephemeral_message(monkeypatch, capsys) -> None: + calls = [] + + def run_shared(arguments, payload): + calls.append(arguments) + if arguments == ["user-prompt"]: + return 0, json.dumps( + {"hookSpecificOutput": {"additionalContext": "Earlier repository context."}} + ) + return 0, "" + + monkeypatch.setattr(adapter, "_run_shared", run_shared) + monkeypatch.setattr(sys, "argv", ["adapter.py", "PreInvocation"]) + monkeypatch.setattr(sys, "stdin", io.StringIO('{"invocationNum": 0, "workspacePaths": ["/repo"]}')) + + assert adapter.main() == 0 + assert calls == [["session-start"], ["user-prompt"]] + assert json.loads(capsys.readouterr().out) == { + "injectSteps": [{"ephemeralMessage": "Earlier repository context."}] + } + + +def test_native_antigravity_bundle_uses_supported_events(tmp_path: Path) -> None: + root = build("antigravity", "native", tmp_path / "antigravity") + + manifest = json.loads((root / "plugin.json").read_text(encoding="utf-8")) + hooks = json.loads((root / "hooks.json").read_text(encoding="utf-8"))["mem0"] + assert manifest["$schema"] == "https://antigravity.google/schemas/v1/plugin.json" + assert set(hooks) == {"PreInvocation", "PostToolUse", "Stop"} + assert (root / "mcp_config.json").is_file() + assert (root / "agents" / "sidekick" / "agent.md").is_file() + assert not any(path.is_symlink() for path in root.rglob("*")) + + +def test_stop_captures_later_prompts_once(tmp_path): + transcript = tmp_path / "transcript.jsonl" + store = adapter.hook_runner.EvidenceStore(tmp_path / "evidence.sqlite3") + payload = {"session_id": "s1", "cwd": str(tmp_path), "transcript_path": str(transcript)} + turns = [] + try: + for prompt, answer in [("First question", "First answer"), ("Next question", "Next answer")]: + turns.extend([ + {"type": "USER_INPUT", "status": "DONE", "content": prompt}, + {"type": "PLANNER_RESPONSE", "source": "MODEL", "status": "DONE", "content": answer}, + ]) + transcript.write_text(''.join(json.dumps(row) + '\n' for row in turns)) + adapter._record_stop(store, payload) + adapter._record_stop(store, payload) + rows = store.conn.execute("SELECT payload_json FROM events WHERE kind = 'assistant_stop' ORDER BY id").fetchall() + messages = [message for row in rows for message in json.loads(row[0])["transcript_messages"]] + assert [m["content"] for m in messages] == ["First question", "First answer", "Next question", "Next answer"] + finally: + store.close() diff --git a/integrations/claude-code-plugin/.claude-plugin/plugin.json b/integrations/claude-code-plugin/.claude-plugin/plugin.json index 3fd2ceb02..6f214cc14 100644 --- a/integrations/claude-code-plugin/.claude-plugin/plugin.json +++ b/integrations/claude-code-plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "mem0", - "version": "0.3.0", + "version": "0.3.1", "description": "Cross-session memory and token savings for coding agents.", "author": { "name": "Mem0" diff --git a/integrations/claude-code-plugin/.gitignore b/integrations/claude-code-plugin/.gitignore deleted file mode 100644 index 7f1a2fd19..000000000 --- a/integrations/claude-code-plugin/.gitignore +++ /dev/null @@ -1,11 +0,0 @@ -__pycache__/ -*.py[cod] -.pytest_cache/ -.ruff_cache/ -.venv/ -*.sqlite3 -*.sqlite3-shm -*.sqlite3-wal -flush-worker.log -plugin-errors.log -pending/ diff --git a/integrations/claude-code-plugin/README.md b/integrations/claude-code-plugin/README.md index 87c535502..ae80910fc 100644 --- a/integrations/claude-code-plugin/README.md +++ b/integrations/claude-code-plugin/README.md @@ -34,10 +34,11 @@ To remove: claude plugin uninstall mem0@mem0-plugins ``` -For local development, load the current checkout directly: +For local development, verify and load the self-contained plugin directory: ```bash -claude --plugin-dir . +python3 integrations/agent-plugin-core/build/build.py claude-code --kind native --check +claude --plugin-dir integrations/claude-code-plugin ``` ## How it works @@ -104,7 +105,11 @@ Categories for `--category`: `project_knowledge`, `decisions_and_constraints`, ` | `dir` | Project memory from the current directory (and children), plus your preferences | | `mine` | Your personal preferences only | -Set the default with the `search_scope` setting or `MEM0_CODE_SEARCH_SCOPE`. Pass `--run-id ` to see only what one specific session recorded. +Set the default with the `search_scope` setting or `MEM0_CODE_SEARCH_SCOPE`. Search spans earlier sessions without a session ID; `run_id` remains internal metadata. + +New Git repository memories use a hash of the remote identity in `agent_id`. Searches also include the previous unhashed ID under the same repository `app_id`, so shared memories remain available after upgrading. Older IDs retain their original limitation: matching owner/repository names on different Git hosts share that legacy namespace. Local folders keep their path-based namespaces. + +Explicit shared-memory deletion with `--include-project-memory` covers both repository IDs. Default deletion preserves shared memories. ## Settings @@ -181,10 +186,10 @@ Breaking update. Memories carry over, most local config does not. ## Development checks -Run from `integrations/claude-code-plugin/`: +Run from the repository root: ```bash -python3 -m pytest tests -q -python3 -m ruff check . -claude plugin validate --strict . +python3 -m pytest integrations/claude-code-plugin/tests -q --ignore=integrations/claude-code-plugin/tests/integration +python3 -m ruff check integrations/agent-plugin-core/python integrations/claude-code-plugin +claude plugin validate --strict integrations/claude-code-plugin ``` diff --git a/integrations/claude-code-plugin/adapters/claude/hook.py b/integrations/claude-code-plugin/adapters/claude/hook.py index 1fcab1177..278cd190d 100644 --- a/integrations/claude-code-plugin/adapters/claude/hook.py +++ b/integrations/claude-code-plugin/adapters/claude/hook.py @@ -3,367 +3,52 @@ from __future__ import annotations -import argparse -import hashlib -import json -import os -import subprocess import sys -import time -import uuid from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "core")) +_here = Path(__file__).resolve() +_bundled_core = _here.parents[2] / "core" +_core_dir = _bundled_core if (_bundled_core / "memory_core.py").is_file() else _bundled_core / "python" +sys.path.insert(0, str(_core_dir)) +sys.path.insert(0, str(_here.parent)) -import telemetry -from memory_core import ( - EvidenceStore, - api_key, - cache_plugin_api_key, - checkpoint_session, - clear_stale_api_key_cache, - data_dir, - detached_process_kwargs, - format_context, - record_session_start, +import telemetry # noqa: E402 +from memory_core import ( # noqa: E402 + configure_harness, record_sidekick_start, record_sidekick_stop, - record_stop, record_tool, - record_user_prompt, - search_memories, ) +from transcript import record_stop # noqa: E402 +import hook_runner # noqa: E402 + +configure_harness("claude-code", data_dir_name="claude-code-plugin", source_tag="claude_code_plugin") +telemetry.init(harness="claude-code", source_tag="CLAUDE_CODE_PLUGIN") -STALE_RUNNING_SECONDS = 300 -PENDING_EXPIRY_SECONDS = 7 * 24 * 60 * 60 -PENDING_LAUNCH_LIMIT = 5 +def _sidekick_start(store, hook_input): + context = record_sidekick_start(store, hook_input) + if context: + return { + "hookSpecificOutput": { + "hookEventName": "SubagentStart", + "additionalContext": context, + }, + } -def read_hook_input() -> dict: - try: - value = json.load(sys.stdin) - return value if isinstance(value, dict) else {} - except (json.JSONDecodeError, OSError): - return {} - - -def first_prompt_memory_output(store: EvidenceStore, hook_input: dict) -> dict: - """Search once before Claude handles the first prompt in a session.""" - repo, session_id, prompt, is_first_prompt = record_user_prompt(store, hook_input) - if not is_first_prompt: - return {} - try: - minimum_query_chars = int(os.environ.get("MEM0_CODE_MIN_QUERY_CHARS", "20")) - except ValueError: - minimum_query_chars = 20 - if len(prompt.strip()) < max(minimum_query_chars, 1): - return {} - result = search_memories( - store, - repo, - session_id, - prompt, - top_k=5, - operation="first-prompt-search", - timeout=2, - ) - if not result.memories: - return {} - context = format_context( - result.memories, - "Mem0 found these relevant memories from earlier work in this repository:", - ) - telemetry.record( - "context_injected", - repo=repo, - session_id=session_id, - trigger="first-prompt", - memory_count=len(result.memories), - context_chars=len(context), - prompt_chars=len(prompt), - ) - return { - "hookSpecificOutput": { - "hookEventName": "UserPromptSubmit", - "additionalContext": context, - }, - } - - -def _launch_handoff(handoff_path: Path) -> bool: - running_path = handoff_path.with_suffix(".running") - try: - handoff_path.replace(running_path) - except OSError: - return False - worker = Path(__file__).resolve().parents[2] / "core" / "flush_worker.py" - log_path = data_dir() / "flush-worker.log" - log_handle = open(log_path, "a", encoding="utf-8") - try: - subprocess.Popen( - [sys.executable, str(worker), str(running_path)], - stdin=subprocess.DEVNULL, - stdout=log_handle, - stderr=log_handle, - close_fds=True, - **detached_process_kwargs(), - ) - finally: - log_handle.close() - return True - - -def recover_pending_handoffs() -> int: - pending_dir = data_dir() / "pending" - pending_dir.mkdir(parents=True, exist_ok=True) - now = time.time() - for running in pending_dir.glob("*.running"): - try: - if now - running.stat().st_mtime > STALE_RUNNING_SECONDS: - running.replace(running.with_suffix(".json")) - except OSError: - continue - recoverable = [] - for handoff in pending_dir.glob("*.json"): - try: - age = now - handoff.stat().st_mtime - except OSError: - continue - if age > PENDING_EXPIRY_SECONDS: - handoff.unlink(missing_ok=True) - continue - recoverable.append((age, handoff)) - recoverable.sort(key=lambda item: item[0], reverse=True) - launched = 0 - for _, handoff in recoverable[:PENDING_LAUNCH_LIMIT]: - launched += int(_launch_handoff(handoff)) - return launched - - -def refresh_pending_handoffs() -> None: - """Hold unsent packets while paused instead of letting them expire.""" - pending_dir = data_dir() / "pending" - if not pending_dir.is_dir(): - return - for pattern in ("*.json", "*.running"): - for handoff in pending_dir.glob(pattern): - try: - os.utime(handoff) - except OSError: - continue - - -def hand_off_flush( - hook_input: dict, reason: str, *, wait_for_inflight: bool = False -) -> None: - """Persist hook input and detach delivery from Claude's shutdown lifecycle.""" - pending_dir = data_dir() / "pending" - pending_dir.mkdir(parents=True, exist_ok=True) - material = ( - f"{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}\0{reason}" - ) - digest = hashlib.sha256(material.encode()).hexdigest()[:24] - handoff_path = pending_dir / f"{digest}-{uuid.uuid4().hex[:8]}.json" - temporary_path = handoff_path.with_suffix(".tmp") - temporary_path.write_text( - json.dumps( - { - "hook_input": hook_input, - "reason": reason, - "wait_for_inflight": wait_for_inflight, - } - ), - encoding="utf-8", - ) - temporary_path.replace(handoff_path) - _launch_handoff(handoff_path) - - -def automatic_flush_enabled() -> bool: - return os.environ.get("MEM0_CODE_AUTO_FLUSH", "true").lower() in { - "1", - "true", - "yes", - "on", - } - - -def schedule_periodic_checkpoint( - store: EvidenceStore, - hook_input: dict, - repo, - session_id: str, -) -> bool: - """Start one background extraction when a complete block is ready.""" - if ( - not automatic_flush_enabled() - or not api_key() - or not store.checkpoint_due(repo.identity, session_id) - ): - return False - if store.prepare_flush(repo, session_id, "periodic") is None: - return False - hand_off_flush(hook_input, "periodic") - return True - - -DEFAULT_IDLE_FLUSH_SECONDS = 300 - - -def _idle_flush_seconds() -> int: - try: - return max( - int(os.environ.get("MEM0_CODE_IDLE_FLUSH_SECONDS", str(DEFAULT_IDLE_FLUSH_SECONDS))), - 0, - ) - except ValueError: - return DEFAULT_IDLE_FLUSH_SECONDS - - -def schedule_idle_flush( - store: EvidenceStore, - hook_input: dict, - repo, - session_id: str, -) -> bool: - """Launch a delayed background flush for sessions that may never end.""" - delay = _idle_flush_seconds() - if delay <= 0 or not automatic_flush_enabled() or not api_key(): - return False - if store.has_inflight_flush(repo.identity, session_id): - return False - if not store.has_unflushed_events(repo.identity, session_id): - return False - pending_dir = data_dir() / "pending" - pending_dir.mkdir(parents=True, exist_ok=True) - material = f"idle\0{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}" - digest = hashlib.sha256(material.encode()).hexdigest()[:24] - for old in pending_dir.glob(f"idle-{digest}*"): - old.unlink(missing_ok=True) - handoff_path = pending_dir / f"idle-{digest}-{uuid.uuid4().hex[:8]}.json" - temporary_path = handoff_path.with_suffix(".tmp") - temporary_path.write_text( - json.dumps({ - "hook_input": hook_input, - "reason": "idle", - "delay_seconds": delay, - }), - encoding="utf-8", - ) - temporary_path.replace(handoff_path) - _launch_handoff(handoff_path) - return True - - -def log_failure(exc: Exception) -> None: - try: - log_path = data_dir() / "plugin-errors.log" - with log_path.open("a", encoding="utf-8") as handle: - handle.write(f"{time.time():.3f} {type(exc).__name__}: {exc}\n") - except OSError: - pass - - -def main() -> int: - parser = argparse.ArgumentParser() - parser.add_argument( - "action", - choices=[ - "session-start", - "user-prompt", - "post-tool", - "post-tool-failure", - "sidekick-start", - "sidekick-stop", - "stop", - "flush", - ], - ) - parser.add_argument("--reason", default="manual") - parser.add_argument("--plugin-data-dir", default="") - args = parser.parse_args() - if args.plugin_data_dir: - os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir - cache_plugin_api_key() - if args.action == "session-start": - clear_stale_api_key_cache() - hook_input = read_hook_input() - store = EvidenceStore() - try: - if store.is_paused(): - if args.action == "session-start": - refresh_pending_handoffs() - telemetry.record("session_start", paused=True) - telemetry.spawn_flush() - return 0 - if args.action == "session-start": - if telemetry.is_first_run(): - telemetry.record("install") - recovered = recover_pending_handoffs() - record_session_start(store, hook_input) - if recovered: - telemetry.record("handoff_recovered", count=recovered) - telemetry.spawn_flush() - elif args.action == "user-prompt": - output = first_prompt_memory_output(store, hook_input) - if output: - print(json.dumps(output)) - elif args.action == "post-tool": - record_tool(store, hook_input) - elif args.action == "post-tool-failure": - record_tool(store, hook_input, failed=True) - elif args.action == "sidekick-start": - context = record_sidekick_start(store, hook_input) - if context: - print( - json.dumps( - { - "hookSpecificOutput": { - "hookEventName": "SubagentStart", - "additionalContext": context, - } - } - ) - ) - elif args.action == "sidekick-stop": - record_sidekick_stop(store, hook_input) - elif args.action == "stop": - repo, session_id = record_stop(store, hook_input) - if not schedule_periodic_checkpoint(store, hook_input, repo, session_id): - schedule_idle_flush(store, hook_input, repo, session_id) - elif args.action == "flush": - automatic = args.reason in {"session-end", "pre-compact"} - if automatic and not automatic_flush_enabled(): - return 0 - if args.reason == "session-end": - # In print mode, SessionEnd can arrive before the Stop hook has - # recorded Claude's final response. Read any remaining visible - # transcript messages before preparing the final extraction. - record_stop(store, hook_input) - if os.environ.get("MEM0_CODE_SYNC_FLUSH") == "1": - print(json.dumps(checkpoint_session(store, hook_input, args.reason))) - else: - session_id = str(hook_input.get("session_id") or "unknown-session") - repo = store.repo_for_session(session_id, hook_input.get("cwd")) - already_running = store.has_inflight_flush(repo.identity, session_id) - if already_running and args.reason == "session-end": - hand_off_flush( - hook_input, args.reason, wait_for_inflight=True - ) - elif not already_running and store.prepare_flush( - repo, session_id, args.reason - ) is not None: - hand_off_flush(hook_input, args.reason) - finally: - store.close() - return 0 +def _sidekick_stop(store, hook_input): + record_sidekick_stop(store, hook_input) if __name__ == "__main__": - try: - raise SystemExit(main()) - except Exception as exc: - # Memory must never prevent the coding agent from continuing. - log_failure(exc) - raise SystemExit(0) + hook_runner.entry_point( + record_stop_fn=record_stop, + extra_actions={ + "post-tool-failure": lambda s, h: record_tool(s, h, failed=True), + "sidekick-start": _sidekick_start, + "sidekick-stop": _sidekick_stop, + }, + data_dir_env="MEM0_CODE_DATA_DIR", + automatic_flush_reasons={"session-end", "pre-compact"}, + ) diff --git a/integrations/claude-code-plugin/adapters/claude/transcript.py b/integrations/claude-code-plugin/adapters/claude/transcript.py new file mode 100644 index 000000000..1d55ccb96 --- /dev/null +++ b/integrations/claude-code-plugin/adapters/claude/transcript.py @@ -0,0 +1,362 @@ +"""Claude Code transcript parsing for Mem0 memory extraction.""" + +from __future__ import annotations + +import html +import json +import re +import sys +from pathlib import Path +from typing import Any + +_here = Path(__file__).resolve() +sys.path.insert(0, str(_here.parents[2] / "core" / "python")) + +from memory_core import ( # noqa: E402 + EvidenceStore, + RepoContext, + _session_id, + redact, +) + + +def _message_content_text(content: Any) -> str: + """Return visible text from one Claude transcript message.""" + if isinstance(content, str): + return redact(content).strip() + if not isinstance(content, list): + return "" + parts = [] + for block in content: + if not isinstance(block, dict) or block.get("type") != "text": + continue + text = redact(block.get("text", "")).strip() + if text: + parts.append(text) + return "\n\n".join(parts) + + +def _transcript_rows( + path: str, offset: int = 0 +) -> tuple[list[dict[str, Any]], int, bool]: + """Parse transcript rows from a byte offset, returning rows, end offset, and whether the offset was honored.""" + if not path: + return [], 0, False + rows = [] + try: + resolved = Path(path).expanduser() + if not 0 <= offset <= resolved.stat().st_size: + offset = 0 + end = offset + with resolved.open("rb") as handle: + handle.seek(offset) + for line in handle: + if not line.endswith(b"\n"): + break + end += len(line) + try: + row = json.loads(line.decode("utf-8", errors="replace")) + except json.JSONDecodeError: + continue + if isinstance(row, dict) and row.get("uuid"): + rows.append(row) + except OSError: + return [], offset, False + return rows, end, offset > 0 + + +def _active_transcript_chain( + rows: list[dict[str, Any]], session_id: str +) -> list[dict[str, Any]]: + """Follow the current Claude conversation branch from its latest record.""" + by_uuid = {str(row["uuid"]): row for row in rows if row.get("uuid")} + leaf = next( + ( + row + for row in reversed(rows) + if not row.get("isSidechain") + and str(row.get("sessionId") or "") == session_id + ), + None, + ) + if leaf is None: + return [] + + chain = [] + seen = set() + current = leaf + while current is not None: + uuid = str(current.get("uuid") or "") + if not uuid or uuid in seen: + break + seen.add(uuid) + chain.append(current) + current = by_uuid.get(str(current.get("parentUuid") or "")) + chain.reverse() + return chain + + +def _human_prompt_text(row: dict[str, Any]) -> str: + if row.get("type") != "user": + return "" + origin = row.get("origin") or {} + if isinstance(origin, dict) and origin.get("kind") not in {None, "human"}: + return "" + message = row.get("message") or {} + content = message.get("content") if isinstance(message, dict) else None + if not isinstance(content, str): + return "" + text = redact(content).strip() + if text.startswith(""): + text = text.removeprefix("").strip() + ignored_prefixes = ( + "", + "", + "", + "", + "", + "", + ) + return "" if text.startswith(ignored_prefixes) else text + + +def _xml_value(text: str, tag: str) -> str: + match = re.search(fr"<{tag}>(.*?)", text, re.DOTALL) + return html.unescape(match.group(1).strip()) if match else "" + + +def _agent_assignment(tool_input: dict[str, Any]) -> str: + prompt = redact(tool_input.get("prompt", "")).strip() + if not prompt: + return "" + agent_type = redact(tool_input.get("subagent_type", "agent")).strip() or "agent" + description = redact(tool_input.get("description", "")).strip() + heading = f"Subagent assignment ({agent_type}" + if description: + heading += f": {description}" + return f"{heading}):\n{prompt}" + + +def _agent_response(tool_input: dict[str, Any], result: str) -> str: + result = redact(result).strip() + if not result or result.startswith("Async agent launched successfully."): + return "" + agent_type = redact(tool_input.get("subagent_type", "agent")).strip() or "agent" + description = redact(tool_input.get("description", "")).strip() + heading = f"Subagent response ({agent_type}" + if description: + heading += f": {description}" + return f"{heading}):\n{result}" + + +def _tool_result_text(block: dict[str, Any]) -> str: + return _message_content_text(block.get("content")) + + +def transcript_extraction_messages( + transcript_path: str, + session_id: str, + *, + previous_leaf_uuid: str = "", + prompt_hint: str = "", + fallback_assistant_message: str = "", + label_final_response: bool = False, + start_offset: int = 0, +) -> tuple[list[dict[str, str]], str, int]: + """Read the meaningful part of the current Claude exchange.""" + rows, end_offset, resumed = _transcript_rows(transcript_path, start_offset) + chain = _active_transcript_chain(rows, session_id) + if not chain: + if resumed: + return [], previous_leaf_uuid, end_offset + fallback = redact(fallback_assistant_message).strip() + return ( + ([{"role": "assistant", "content": f"Main Claude response:\n{fallback}"}] + if fallback + else []), + "", + end_offset, + ) + + leaf_uuid = str(chain[-1].get("uuid") or "") + if previous_leaf_uuid and leaf_uuid == previous_leaf_uuid: + return [], leaf_uuid, end_offset + start = 0 + if previous_leaf_uuid: + for index, row in enumerate(chain): + if str(row.get("uuid") or "") == previous_leaf_uuid: + start = index + 1 + break + else: + previous_leaf_uuid = "" + if not previous_leaf_uuid and not resumed: + prompt_hint = redact(prompt_hint).strip() + candidates = [ + index + for index, row in enumerate(chain) + if _human_prompt_text(row) + and ( + not prompt_hint + or _human_prompt_text(row) == prompt_hint + ) + ] + task_notifications = [ + index + for index, row in enumerate(chain) + if isinstance((row.get("message") or {}).get("content"), str) + and (row.get("message") or {})["content"].startswith("") + ] + if candidates: + start = candidates[-1] + elif task_notifications: + start = task_notifications[-1] + + tool_uses: dict[str, tuple[str, dict[str, Any]]] = {} + for row in chain: + message = row.get("message") or {} + content = message.get("content") if isinstance(message, dict) else None + if not isinstance(content, list): + continue + for block in content: + if not isinstance(block, dict) or block.get("type") != "tool_use": + continue + tool_id = str(block.get("id") or "") + tool_input = block.get("input") or {} + if tool_id and isinstance(tool_input, dict): + tool_uses[tool_id] = (str(block.get("name") or ""), tool_input) + + output: list[dict[str, str]] = [] + + def append(role: str, content: str) -> None: + content = redact(content).strip() + if content: + output.append({"role": role, "content": content}) + + for row in chain[start:]: + message = row.get("message") or {} + if not isinstance(message, dict): + continue + role = str(message.get("role") or "") + content = message.get("content") + + if role == "user" and isinstance(content, str): + if content.startswith(""): + if _xml_value(content, "status") != "completed": + continue + tool_id = _xml_value(content, "tool-use-id") + tool = tool_uses.get(tool_id) + result = _xml_value(content, "result") + if tool and tool[0] == "Agent" and result: + assignment = _agent_assignment(tool[1]) + response = _agent_response(tool[1], result) + append("assistant", assignment) + append("assistant", response) + continue + human = _human_prompt_text(row) + if human: + append("user", human) + continue + + if not isinstance(content, list): + continue + for block in content: + if not isinstance(block, dict): + continue + block_type = block.get("type") + if role == "assistant" and block_type == "text": + append("assistant", str(block.get("text") or "")) + continue + if role != "user" or block_type != "tool_result": + continue + + tool_id = str(block.get("tool_use_id") or "") + tool = tool_uses.get(tool_id) + if not tool: + continue + name, tool_input = tool + result = _tool_result_text(block) + failed = bool(block.get("is_error")) + if name == "Agent" and not failed: + response = _agent_response(tool_input, result) + if response: + append("assistant", _agent_assignment(tool_input)) + append("assistant", response) + elif name == "AskUserQuestion" and result and not failed: + append("user", f"User answers to Claude's questions:\n{result}") + elif name == "ExitPlanMode" and not failed: + plan = redact(tool_input.get("plan", "")).strip() + if plan: + append("assistant", f"Approved implementation plan:\n{plan}") + + fallback = redact(fallback_assistant_message).strip() + if fallback: + labeled = f"Main Claude response:\n{fallback}" + for message in reversed(output): + if message["role"] == "assistant" and message["content"] == fallback: + message["content"] = labeled + break + else: + append("assistant", labeled) + elif label_final_response: + last_message = chain[-1].get("message") or {} + last_content = ( + last_message.get("content") if isinstance(last_message, dict) else None + ) + final_parts = [ + redact(block.get("text", "")).strip() + for block in (last_content if isinstance(last_content, list) else []) + if isinstance(block, dict) + and block.get("type") == "text" + and redact(block.get("text", "")).strip() + ] + for start in range(len(output) - len(final_parts), -1, -1): + candidate = output[start : start + len(final_parts)] + if final_parts and [item["content"] for item in candidate] == final_parts: + candidate[0]["content"] = ( + f"Main Claude response:\n{candidate[0]['content']}" + ) + break + return output, leaf_uuid, end_offset + + +def record_stop( + store: EvidenceStore, hook_input: dict[str, Any] +) -> tuple[RepoContext, str]: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + message = redact(hook_input.get("last_assistant_message", "")).strip() + previous_stop = store.latest_event_payload( + repo.identity, session_id, "assistant_stop" + ) + latest_prompt = store.latest_event_payload(repo.identity, session_id, "user_prompt") + transcript_path = str(hook_input.get("transcript_path") or "") + previous_offset = previous_stop.get("transcript_offset") + start_offset = ( + previous_offset + if isinstance(previous_offset, int) + and str(previous_stop.get("transcript_path") or "") == transcript_path + else 0 + ) + transcript_messages, leaf_uuid, end_offset = transcript_extraction_messages( + transcript_path, + session_id, + previous_leaf_uuid=str(previous_stop.get("transcript_leaf_uuid") or ""), + prompt_hint=str(latest_prompt.get("text") or ""), + fallback_assistant_message=message, + label_final_response=True, + start_offset=start_offset, + ) + if transcript_messages: + payload: dict[str, Any] = { + "text": message, + "transcript_messages": transcript_messages, + } + if leaf_uuid: + payload["transcript_leaf_uuid"] = leaf_uuid + if transcript_path: + payload["transcript_path"] = transcript_path + payload["transcript_offset"] = end_offset + store.record_event( + repo, session_id, "assistant_stop", payload + ) + return repo, session_id diff --git a/integrations/claude-code-plugin/core/flush_worker.py b/integrations/claude-code-plugin/core/flush_worker.py index c60fa82a5..6f7b9ecb2 100644 --- a/integrations/claude-code-plugin/core/flush_worker.py +++ b/integrations/claude-code-plugin/core/flush_worker.py @@ -17,7 +17,7 @@ import telemetry from memory_core import ( EvidenceStore, checkpoint_session, - record_stop, + configure_harness, touch_handoff_heartbeat, ) @@ -27,15 +27,28 @@ def main() -> int: return 2 handoff_path = Path(sys.argv[1]) os.environ["MEM0_CODE_HANDOFF_PATH"] = str(handoff_path) + harness = os.environ.get("MEM0_PLUGIN_HARNESS") + if harness: + source_tag = os.environ.get("MEM0_PLUGIN_SOURCE_TAG", "") + configure_harness( + harness, + env_prefix=os.environ.get("MEM0_PLUGIN_ENV_PREFIX", ""), + data_dir_name=os.environ.get("MEM0_PLUGIN_DATA_DIR_NAME", ""), + source_tag=source_tag, + ) + telemetry.init(harness=harness, source_tag=source_tag.upper()) completed = False try: payload = json.loads(handoff_path.read_text(encoding="utf-8")) delay = float(payload.get("delay_seconds") or 0) if delay > 0: payload.pop("delay_seconds", None) - handoff_path.write_text( - json.dumps(payload), encoding="utf-8" - ) + temporary = handoff_path.with_suffix(f".{os.getpid()}.tmp") + try: + temporary.write_text(json.dumps(payload), encoding="utf-8") + temporary.replace(handoff_path) + finally: + temporary.unlink(missing_ok=True) time.sleep(delay) if not handoff_path.exists(): return 0 @@ -58,8 +71,7 @@ def main() -> int: ): touch_handoff_heartbeat() time.sleep(0.25) - if reason == "session-end": - record_stop(store, hook_input) + # Hooks capture the conversation before handoff; the worker only flushes it. result = checkpoint_session(store, hook_input, reason) print(json.dumps(result, sort_keys=True), flush=True) completed = result.get("status") in { diff --git a/integrations/claude-code-plugin/core/hook_runner.py b/integrations/claude-code-plugin/core/hook_runner.py new file mode 100644 index 000000000..2a2adb557 --- /dev/null +++ b/integrations/claude-code-plugin/core/hook_runner.py @@ -0,0 +1,372 @@ +"""Shared hook orchestration for all Mem0 agent plugins.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import subprocess +import sys +import time +import uuid +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + _session_id, + api_key, + bounded, + cache_plugin_api_key, + checkpoint_session, + clear_stale_api_key_cache, + configure_harness, + data_dir, + detached_process_kwargs, + format_context, + harness_config, + record_session_start, + record_tool, + record_user_prompt, + redact, + search_memories, +) + +STALE_RUNNING_SECONDS = 300 +PENDING_EXPIRY_SECONDS = 7 * 24 * 60 * 60 +PENDING_LAUNCH_LIMIT = 5 +DEFAULT_IDLE_FLUSH_SECONDS = 300 + +_core_dir: Path = Path(__file__).resolve().parent + + +def read_hook_input() -> dict: + try: + value = json.load(sys.stdin) + return value if isinstance(value, dict) else {} + except (json.JSONDecodeError, OSError): + return {} + + +def default_record_stop(store: EvidenceStore, hook_input: dict): + """Record the assistant's response without transcript parsing.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + message = redact(hook_input.get("last_assistant_message", "")).strip() + if message: + store.record_assistant_response(repo, session_id, message) + return repo, session_id + + +def first_prompt_memory_output(store: EvidenceStore, hook_input: dict) -> dict: + """Search once before the agent handles the first prompt in a session.""" + repo, session_id, prompt, is_first_prompt = record_user_prompt(store, hook_input) + if not is_first_prompt: + return {} + try: + minimum_query_chars = int(os.environ.get("MEM0_CODE_MIN_QUERY_CHARS", "20")) + except ValueError: + minimum_query_chars = 20 + if len(prompt.strip()) < max(minimum_query_chars, 1): + return {} + result = search_memories( + store, repo, session_id, bounded(prompt, 6000), + top_k=5, operation="first-prompt-search", timeout=2, + ) + if not result.memories: + return {} + context = format_context( + result.memories, + "Mem0 found these relevant memories from earlier work in this repository:", + ) + telemetry.record( + "context_injected", + repo=repo, session_id=session_id, trigger="first-prompt", + memory_count=len(result.memories), context_chars=len(context), + prompt_chars=len(prompt), + ) + return { + "hookSpecificOutput": { + "hookEventName": "UserPromptSubmit", + "additionalContext": context, + }, + } + + +def _launch_handoff(handoff_path: Path) -> bool: + running_path = handoff_path.with_suffix(".running") + try: + handoff_path.replace(running_path) + except OSError: + return False + worker = _core_dir / "flush_worker.py" + log_path = data_dir() / "flush-worker.log" + log_handle = open(log_path, "a", encoding="utf-8") + harness = harness_config() + child_env = os.environ.copy() + child_env.update( + { + "MEM0_CODE_DATA_DIR": str(data_dir()), + "MEM0_PLUGIN_HARNESS": harness["name"], + "MEM0_PLUGIN_ENV_PREFIX": harness["env_prefix"], + "MEM0_PLUGIN_DATA_DIR_NAME": harness["data_dir_name"], + "MEM0_PLUGIN_SOURCE_TAG": harness["source_tag"], + } + ) + try: + subprocess.Popen( + [sys.executable, str(worker), str(running_path)], + stdin=subprocess.DEVNULL, + stdout=log_handle, stderr=log_handle, + close_fds=True, + env=child_env, + **detached_process_kwargs(), + ) + finally: + log_handle.close() + return True + + +def recover_pending_handoffs() -> int: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + now = time.time() + for running in pending_dir.glob("*.running"): + try: + if now - running.stat().st_mtime > STALE_RUNNING_SECONDS: + running.replace(running.with_suffix(".json")) + except OSError: + continue + recoverable = [] + for handoff in pending_dir.glob("*.json"): + try: + age = now - handoff.stat().st_mtime + except OSError: + continue + if age > PENDING_EXPIRY_SECONDS: + handoff.unlink(missing_ok=True) + continue + recoverable.append((age, handoff)) + recoverable.sort(key=lambda item: item[0], reverse=True) + launched = 0 + for _, handoff in recoverable[:PENDING_LAUNCH_LIMIT]: + launched += int(_launch_handoff(handoff)) + return launched + + +def refresh_pending_handoffs() -> None: + pending_dir = data_dir() / "pending" + if not pending_dir.is_dir(): + return + for pattern in ("*.json", "*.running"): + for handoff in pending_dir.glob(pattern): + try: + os.utime(handoff) + except OSError: + continue + + +def hand_off_flush( + hook_input: dict, reason: str, *, wait_for_inflight: bool = False, +) -> None: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = ( + f"{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}\0{reason}" + ) + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + handoff_path = pending_dir / f"{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": reason, + "wait_for_inflight": wait_for_inflight, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + + +def automatic_flush_enabled() -> bool: + return os.environ.get("MEM0_CODE_AUTO_FLUSH", "true").lower() in { + "1", "true", "yes", "on", + } + + +def schedule_periodic_checkpoint( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + if ( + not automatic_flush_enabled() + or not api_key() + or not store.checkpoint_due(repo.identity, session_id) + ): + return False + if store.prepare_flush(repo, session_id, "periodic") is None: + return False + hand_off_flush(hook_input, "periodic") + return True + + +def _idle_flush_seconds() -> int: + try: + return max( + int(os.environ.get("MEM0_CODE_IDLE_FLUSH_SECONDS", str(DEFAULT_IDLE_FLUSH_SECONDS))), + 0, + ) + except ValueError: + return DEFAULT_IDLE_FLUSH_SECONDS + + +def schedule_idle_flush( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + delay = _idle_flush_seconds() + if delay <= 0 or not automatic_flush_enabled() or not api_key(): + return False + if store.has_inflight_flush(repo.identity, session_id): + return False + if not store.has_unflushed_events(repo.identity, session_id): + return False + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = f"idle\0{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}" + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + for old in pending_dir.glob(f"idle-{digest}*"): + old.unlink(missing_ok=True) + handoff_path = pending_dir / f"idle-{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": "idle", + "delay_seconds": delay, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + return True + + +def log_failure(exc: Exception) -> None: + try: + log_path = data_dir() / "plugin-errors.log" + with log_path.open("a", encoding="utf-8") as handle: + handle.write(f"{time.time():.3f} {type(exc).__name__}: {exc}\n") + except OSError: + pass + + +def run( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> int: + if record_stop_fn is None: + record_stop_fn = default_record_stop + if automatic_flush_reasons is None: + automatic_flush_reasons = {"session-end"} + + base_actions = ["session-start", "user-prompt", "post-tool", "stop", "flush"] + all_actions = base_actions + list((extra_actions or {}).keys()) + + parser = argparse.ArgumentParser() + parser.add_argument("action", choices=all_actions) + parser.add_argument("--reason", default="manual") + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + args = parser.parse_args() + + if args.harness: + configure_harness(args.harness) + telemetry.init(harness=args.harness) + + if args.plugin_data_dir: + os.environ[data_dir_env] = args.plugin_data_dir + + cache_plugin_api_key() + if args.action == "session-start": + clear_stale_api_key_cache() + + hook_input = read_hook_input() + store = EvidenceStore() + try: + if store.is_paused(): + if args.action == "session-start": + refresh_pending_handoffs() + telemetry.record("session_start", paused=True) + telemetry.spawn_flush() + return 0 + + if args.action == "session-start": + if telemetry.is_first_run(): + telemetry.record("install") + recovered = recover_pending_handoffs() + record_session_start(store, hook_input) + if recovered: + telemetry.record("handoff_recovered", count=recovered) + telemetry.spawn_flush() + elif args.action == "user-prompt": + output = first_prompt_memory_output(store, hook_input) + if output: + print(json.dumps(output)) + elif args.action == "post-tool": + record_tool(store, hook_input) + elif args.action == "stop": + repo, session_id = record_stop_fn(store, hook_input) + if not schedule_periodic_checkpoint(store, hook_input, repo, session_id): + schedule_idle_flush(store, hook_input, repo, session_id) + elif args.action == "flush": + automatic = args.reason in automatic_flush_reasons + if automatic and not automatic_flush_enabled(): + return 0 + if args.reason == "session-end": + record_stop_fn(store, hook_input) + if os.environ.get("MEM0_CODE_SYNC_FLUSH") == "1": + print(json.dumps(checkpoint_session(store, hook_input, args.reason))) + else: + session_id = str(hook_input.get("session_id") or "unknown-session") + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + already_running = store.has_inflight_flush(repo.identity, session_id) + if already_running and args.reason == "session-end": + hand_off_flush(hook_input, args.reason, wait_for_inflight=True) + elif not already_running and store.prepare_flush( + repo, session_id, args.reason, + ) is not None: + hand_off_flush(hook_input, args.reason) + elif extra_actions and args.action in extra_actions: + result = extra_actions[args.action](store, hook_input) + if result: + print(json.dumps(result)) + finally: + store.close() + return 0 + + +def entry_point( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> None: + try: + raise SystemExit(run( + record_stop_fn=record_stop_fn, + extra_actions=extra_actions, + data_dir_env=data_dir_env, + automatic_flush_reasons=automatic_flush_reasons, + )) + except Exception as exc: + log_failure(exc) + raise SystemExit(0) + + +if __name__ == "__main__": + entry_point() diff --git a/integrations/claude-code-plugin/core/mcp_server.py b/integrations/claude-code-plugin/core/mcp_server.py index 2a9a37c07..036fbbdc9 100644 --- a/integrations/claude-code-plugin/core/mcp_server.py +++ b/integrations/claude-code-plugin/core/mcp_server.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Expose Mem0's memory search as one local Claude Code tool.""" +"""Expose Mem0's memory search as one local coding-agent tool.""" from __future__ import annotations @@ -11,13 +11,13 @@ from typing import Any import telemetry from memory_core import ( CODING_MEMORY_CATEGORY_NAMES, + PLUGIN_VERSION, + SEARCH_SCOPES, format_search_result, resolve_repo, - SEARCH_SCOPES, search_memories, ) - PROTOCOL_VERSION = "2024-11-05" TOOL_NAME = "search_memories" TOOL_DESCRIPTION = ( @@ -30,8 +30,7 @@ TOOL_DESCRIPTION = ( "invocation works. The scope argument changes what is searched: 'repo' " "(default) is the whole repository's shared memory plus your own " "preferences, 'dir' narrows the shared part to the directory you are " - "working in, and 'mine' is your preferences alone. Pass run_id to look " - "at one earlier Claude Code session only." + "working in, and 'mine' is your preferences alone." ) TOOL_SCHEMA = { "type": "object", @@ -65,8 +64,10 @@ TOOL_SCHEMA = { "run_id": { "type": "string", "minLength": 1, - "maxLength": 200, - "description": "Optional Claude Code session ID. Restricts the search to memories written from that session.", + "description": ( + "Optional coding-agent session ID. With any scope, restricts results to memories " + "saved in that session. Omit to recall memories across sessions." + ), }, }, "required": ["query"], @@ -110,16 +111,17 @@ def _validate_arguments( raise ToolInputError(f"scope must be one of {list(SEARCH_SCOPES)}.") run_id = arguments.get("run_id") - if run_id is not None and ( - not isinstance(run_id, str) or not run_id.strip() or len(run_id) > 200 - ): - raise ToolInputError("run_id must be a non-empty string of at most 200 characters.") + if run_id is not None: + if not isinstance(run_id, str) or not run_id.strip(): + raise ToolInputError("run_id must be a non-empty string.") + run_id = run_id.strip() + return query, top_k, category, scope, run_id -def call_search_memories(arguments: Any) -> str: +def call_search_memories(arguments: Any, cwd: str | None = None) -> str: query, top_k, category, scope, run_id = _validate_arguments(arguments) - repo = resolve_repo(os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + repo = resolve_repo(cwd or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) result = search_memories( None, repo, @@ -134,6 +136,19 @@ def call_search_memories(arguments: Any) -> str: return format_search_result(result) +def _workspace_cwd(params: dict[str, Any]) -> str | None: + meta = params.get("_meta") + if not isinstance(meta, dict): + return None + metadata = meta.get("x-codex-turn-metadata") + if not isinstance(metadata, dict): + return None + workspaces = metadata.get("workspaces") or {} + if isinstance(workspaces, dict): + return next((path for path in workspaces if isinstance(path, str) and path), None) + return None + + def _tool_response(text: str, *, is_error: bool = False) -> dict[str, Any]: return { "content": [{"type": "text", "text": text}], @@ -157,7 +172,7 @@ def handle_request(message: Any) -> dict[str, Any] | None: "result": { "protocolVersion": requested or PROTOCOL_VERSION, "capabilities": {"tools": {"listChanged": False}}, - "serverInfo": {"name": "mem0", "version": "0.3.0"}, + "serverInfo": {"name": "mem0", "version": PLUGIN_VERSION}, }, } if method == "ping": @@ -187,7 +202,9 @@ def handle_request(message: Any) -> dict[str, Any] | None: result = _tool_response("Unknown Mem0 tool.", is_error=True) else: try: - result = _tool_response(call_search_memories(params.get("arguments"))) + result = _tool_response( + call_search_memories(params.get("arguments"), _workspace_cwd(params)) + ) except ToolInputError as exc: result = _tool_response(str(exc), is_error=True) except Exception: diff --git a/integrations/claude-code-plugin/core/memory_cli.py b/integrations/claude-code-plugin/core/memory_cli.py index eb1b595b9..595729feb 100644 --- a/integrations/claude-code-plugin/core/memory_cli.py +++ b/integrations/claude-code-plugin/core/memory_cli.py @@ -14,6 +14,7 @@ from memory_core import ( data_dir, doctor, forget_remote_repo, + configure_harness, resolve_repo, user_id, ) @@ -60,6 +61,7 @@ def _print_status(value: dict) -> None: def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") subparsers = parser.add_subparsers(dest="command", required=True) status = subparsers.add_parser("status") @@ -77,6 +79,10 @@ def main() -> int: forget.add_argument("--include-project-memory", action="store_true") args = parser.parse_args() + if args.harness: + source_tag = f"{args.harness.replace('-', '_')}_plugin" + configure_harness(args.harness, source_tag=source_tag) + telemetry.init(harness=args.harness, source_tag=source_tag.upper()) if args.plugin_data_dir: os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir store = EvidenceStore() diff --git a/integrations/claude-code-plugin/core/memory_core.py b/integrations/claude-code-plugin/core/memory_core.py index b9d9267b7..cf71196b8 100644 --- a/integrations/claude-code-plugin/core/memory_core.py +++ b/integrations/claude-code-plugin/core/memory_core.py @@ -1,16 +1,15 @@ #!/usr/bin/env python3 -"""Save useful coding memories and search them in later Claude Code sessions. +"""Shared core for Mem0 agent plugins. -Hooks record small session details locally. When Claude compacts or ends the -session, Mem0 sends the useful parts of the session to Mem0 so it can create -memories. Claude can search those memories during later work in the repository. +Hooks record small session details locally. When the agent compacts or ends the +session, Mem0 sends the useful parts to the platform so it can create memories. +The agent can search those memories during later work in the repository. """ from __future__ import annotations import functools import hashlib -import html import json import math import os @@ -20,20 +19,46 @@ import subprocess import sys import time import urllib.error -import urllib.request import urllib.parse +import urllib.request from dataclasses import dataclass -from datetime import timezone, datetime +from datetime import datetime, timezone from pathlib import Path from typing import Any, Iterable import telemetry - DEFAULT_API_URL = "https://api.mem0.ai" -PLUGIN_VERSION = "0.3.0" -MAX_PROMPT_CHARS = 6000 -MAX_ASSISTANT_CHARS = 6000 +PLUGIN_VERSION = "0.3.1" + +_harness_name: str = "generic" +_harness_env_prefix: str = "MEM0_PLUGIN" +_harness_data_dir_name: str = "mem0-plugin" +_harness_source_tag: str = "mem0_plugin" + + +def configure_harness( + name: str, + env_prefix: str = "", + data_dir_name: str = "", + source_tag: str = "", +) -> None: + global _harness_name, _harness_env_prefix, _harness_data_dir_name, _harness_source_tag + _harness_name = name + _harness_env_prefix = env_prefix or f"MEM0_{name.upper().replace('-', '_')}" + _harness_data_dir_name = data_dir_name or f"{name}-plugin" + _harness_source_tag = source_tag or f"{name.replace('-', '_')}_plugin" + + +def harness_config() -> dict[str, str]: + return { + "name": _harness_name, + "env_prefix": _harness_env_prefix, + "data_dir_name": _harness_data_dir_name, + "source_tag": _harness_source_tag, + } + + MAX_COMMAND_CHARS = 2000 MAX_RESULT_CHARS = 2500 MAX_EPISODE_CHARS = 12000 @@ -52,7 +77,7 @@ A completed change should produce one memory explaining the resulting behavior, A command that failed and was then made to work should produce one memory naming the failing invocation, the error it returned, and the invocation that succeeded. Do not save one-off errors caused by an edit still in progress, transient network failures, or anything a rerun would fix on its own. -Use Claude's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or Claude completed them. Treat subagent responses as supporting repository evidence, not as decisions. +Use the current coding agent's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or the coding agent completed them. Treat subagent responses as supporting repository evidence, not as decisions. Write about the repository, not the user, assistant, session, or task. Do not save personal preferences. Do not save a memory that only states which repository, branch, or directory the session worked in. Do not include test results, documentation updates, release notes, or temporary state. @@ -130,6 +155,11 @@ SECRET_PATTERNS = [ r"-----BEGIN [^-]*PRIVATE KEY-----.*?-----END [^-]*PRIVATE KEY-----", re.DOTALL, ), + re.compile( + r'(?i)("(?:api[_-]?key|password|secret(?:[_-]?access[_-]?key)?' + r'|(?:access|refresh|session)[_-]?token|token|authorization|credential' + r')"\s*:\s*")(?:\\.|[^"\\])*' + ), ] @@ -213,13 +243,21 @@ def directory_chain(repo: RepoContext) -> list[str]: return ["/".join(parts[: index + 1]) for index in range(len(parts))] +def _shared_project_ids(repo: RepoContext) -> list[str]: + """Current and pre-upgrade namespaces, shared by recall and explicit deletion.""" + if not repo.identity.startswith("local:") and repo.project_id != repo.app_id: + return [repo.project_id, repo.app_id] + return [repo.project_id] + + def _search_filters(user: str, repo: RepoContext, scope: str) -> dict[str, Any]: """Build the scope filter: app_id scopes to the repo, then union shared and personal lanes.""" app_scope = {"app_id": repo.app_id} mine = {"AND": [{"user_id": user}, app_scope]} if scope == "mine": return mine - shared: dict[str, Any] = {"AND": [{"agent_id": repo.project_id}, app_scope]} + projects = [{"AND": [{"agent_id": project_id}, app_scope]} for project_id in _shared_project_ids(repo)] + shared: dict[str, Any] = projects[0] if len(projects) == 1 else {"OR": projects} if scope == "dir" and repo.directory: shared = {"AND": [shared, {"metadata": {"dirs": {"contains": repo.directory}}}]} return {"OR": [shared, mine]} @@ -297,9 +335,9 @@ class RepoContext: def _project_id(root: str, identity: str, app_id: str) -> str: - """The shared namespace: the repository, or a folder path hashed so same-named folders stay apart.""" + """The shared namespace: includes a host hash so repos with the same owner/name on different hosts stay apart.""" if not identity.startswith("local:"): - return app_id + return f"{app_id}-{hashlib.sha256(identity.encode()).hexdigest()[:10]}" return f"local-{app_id}-{hashlib.sha256(root.encode()).hexdigest()[:10]}" @@ -345,8 +383,8 @@ def resolve_repo(cwd: str | None) -> RepoContext: def api_key() -> str: configured = ( os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") - # Compatibility with the pre-marketplace development harness. or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") or "" ).strip() @@ -359,9 +397,10 @@ def api_key() -> str: def cache_plugin_api_key() -> bool: - """Bridge Claude's hook-only sensitive config into plugin-owned storage.""" + """Bridge host's hook-only sensitive config into plugin-owned storage.""" configured = ( - os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") or "" ).strip() @@ -394,6 +433,7 @@ def clear_stale_api_key_cache() -> bool: """Drop the cached key file once every configured key source is gone.""" configured = ( os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") or "" @@ -411,7 +451,7 @@ def clear_stale_api_key_cache() -> bool: def detached_process_kwargs(platform: str | None = None) -> dict: - """Keep a spawned worker alive after Claude Code exits, on POSIX and Windows.""" + """Keep a spawned worker alive after the coding agent exits, on POSIX and Windows.""" if (platform or sys.platform) == "win32": return { "creationflags": subprocess.DETACHED_PROCESS @@ -422,7 +462,8 @@ def detached_process_kwargs(platform: str | None = None) -> dict: def _plugin_option(name: str, fallback: str = "") -> str: return ( - os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") + os.environ.get(f"PLUGIN_OPTION_{name.upper()}") + or os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") or os.environ.get(fallback) or "" ).strip() @@ -440,11 +481,14 @@ def user_id() -> str: def data_dir() -> Path: - configured = os.environ.get("MEM0_CODE_DATA_DIR") or os.environ.get( - "CLAUDE_PLUGIN_DATA" + configured = ( + os.environ.get("MEM0_CODE_DATA_DIR") + or os.environ.get("MEM0_PLUGIN_DATA_DIR") + or os.environ.get("PLUGIN_DATA") + or os.environ.get("CLAUDE_PLUGIN_DATA") ) return ( - Path(configured).expanduser() if configured else Path.home() / ".mem0" / "claude-code-plugin" + Path(configured).expanduser() if configured else Path.home() / ".mem0" / _harness_data_dir_name ) @@ -464,316 +508,11 @@ def _int_option(name: str, fallback: str, default: int) -> int: -def _message_content_text(content: Any) -> str: - """Return visible text from one Claude transcript message.""" - if isinstance(content, str): - return redact(content).strip() - if not isinstance(content, list): - return "" - parts = [] - for block in content: - if not isinstance(block, dict) or block.get("type") != "text": - continue - text = redact(block.get("text", "")).strip() - if text: - parts.append(text) - return "\n\n".join(parts) - - -def _transcript_rows( - path: str, offset: int = 0 -) -> tuple[list[dict[str, Any]], int, bool]: - """Parse transcript rows from a byte offset, returning rows, end offset, and whether the offset was honored.""" - if not path: - return [], 0, False - rows = [] - try: - resolved = Path(path).expanduser() - if not 0 <= offset <= resolved.stat().st_size: - offset = 0 - end = offset - with resolved.open("rb") as handle: - handle.seek(offset) - for line in handle: - if not line.endswith(b"\n"): - break - end += len(line) - try: - row = json.loads(line.decode("utf-8", errors="replace")) - except json.JSONDecodeError: - continue - if isinstance(row, dict) and row.get("uuid"): - rows.append(row) - except OSError: - return [], offset, False - return rows, end, offset > 0 - - -def _active_transcript_chain( - rows: list[dict[str, Any]], session_id: str -) -> list[dict[str, Any]]: - """Follow the current Claude conversation branch from its latest record.""" - by_uuid = {str(row["uuid"]): row for row in rows if row.get("uuid")} - leaf = next( - ( - row - for row in reversed(rows) - if not row.get("isSidechain") - and str(row.get("sessionId") or "") == session_id - ), - None, - ) - if leaf is None: - return [] - - chain = [] - seen = set() - current = leaf - while current is not None: - uuid = str(current.get("uuid") or "") - if not uuid or uuid in seen: - break - seen.add(uuid) - chain.append(current) - current = by_uuid.get(str(current.get("parentUuid") or "")) - chain.reverse() - return chain - - -def _human_prompt_text(row: dict[str, Any]) -> str: - if row.get("type") != "user": - return "" - origin = row.get("origin") or {} - if isinstance(origin, dict) and origin.get("kind") not in {None, "human"}: - return "" - message = row.get("message") or {} - content = message.get("content") if isinstance(message, dict) else None - if not isinstance(content, str): - return "" - text = redact(content).strip() - if text.startswith(""): - text = text.removeprefix("").strip() - ignored_prefixes = ( - "", - "", - "", - "", - "", - "", - ) - return "" if text.startswith(ignored_prefixes) else text - - -def _xml_value(text: str, tag: str) -> str: - match = re.search(fr"<{tag}>(.*?)", text, re.DOTALL) - return html.unescape(match.group(1).strip()) if match else "" - - -def _agent_assignment(tool_input: dict[str, Any]) -> str: - prompt = redact(tool_input.get("prompt", "")).strip() - if not prompt: - return "" - agent_type = redact(tool_input.get("subagent_type", "agent")).strip() or "agent" - description = redact(tool_input.get("description", "")).strip() - heading = f"Subagent assignment ({agent_type}" - if description: - heading += f": {description}" - return f"{heading}):\n{prompt}" - - -def _agent_response(tool_input: dict[str, Any], result: str) -> str: - result = redact(result).strip() - if not result or result.startswith("Async agent launched successfully."): - return "" - agent_type = redact(tool_input.get("subagent_type", "agent")).strip() or "agent" - description = redact(tool_input.get("description", "")).strip() - heading = f"Subagent response ({agent_type}" - if description: - heading += f": {description}" - return f"{heading}):\n{result}" - - -def _tool_result_text(block: dict[str, Any]) -> str: - return _message_content_text(block.get("content")) - - -def transcript_extraction_messages( - transcript_path: str, - session_id: str, - *, - previous_leaf_uuid: str = "", - prompt_hint: str = "", - fallback_assistant_message: str = "", - label_final_response: bool = False, - start_offset: int = 0, -) -> tuple[list[dict[str, str]], str, int]: - """Read the meaningful part of the current Claude exchange. - - The returned messages contain human prompts, visible Claude text, accepted - plans, answers collected through AskUserQuestion, and completed native - subagent assignments and responses. Raw tool output and hidden reasoning - are deliberately excluded. - """ - rows, end_offset, resumed = _transcript_rows(transcript_path, start_offset) - chain = _active_transcript_chain(rows, session_id) - if not chain: - if resumed: - return [], previous_leaf_uuid, end_offset - fallback = redact(fallback_assistant_message).strip() - return ( - ([{"role": "assistant", "content": f"Main Claude response:\n{fallback}"}] - if fallback - else []), - "", - end_offset, - ) - - leaf_uuid = str(chain[-1].get("uuid") or "") - if previous_leaf_uuid and leaf_uuid == previous_leaf_uuid: - return [], leaf_uuid, end_offset - start = 0 - if previous_leaf_uuid: - for index, row in enumerate(chain): - if str(row.get("uuid") or "") == previous_leaf_uuid: - start = index + 1 - break - else: - previous_leaf_uuid = "" - if not previous_leaf_uuid and not resumed: - prompt_hint = redact(prompt_hint).strip() - candidates = [ - index - for index, row in enumerate(chain) - if _human_prompt_text(row) - and ( - not prompt_hint - or _human_prompt_text(row) == prompt_hint - ) - ] - task_notifications = [ - index - for index, row in enumerate(chain) - if isinstance((row.get("message") or {}).get("content"), str) - and (row.get("message") or {})["content"].startswith("") - ] - if candidates: - start = candidates[-1] - elif task_notifications: - start = task_notifications[-1] - - tool_uses: dict[str, tuple[str, dict[str, Any]]] = {} - for row in chain: - message = row.get("message") or {} - content = message.get("content") if isinstance(message, dict) else None - if not isinstance(content, list): - continue - for block in content: - if not isinstance(block, dict) or block.get("type") != "tool_use": - continue - tool_id = str(block.get("id") or "") - tool_input = block.get("input") or {} - if tool_id and isinstance(tool_input, dict): - tool_uses[tool_id] = (str(block.get("name") or ""), tool_input) - - output: list[dict[str, str]] = [] - - def append(role: str, content: str) -> None: - content = redact(content).strip() - if content: - output.append({"role": role, "content": content}) - - for row in chain[start:]: - message = row.get("message") or {} - if not isinstance(message, dict): - continue - role = str(message.get("role") or "") - content = message.get("content") - - if role == "user" and isinstance(content, str): - if content.startswith(""): - if _xml_value(content, "status") != "completed": - continue - tool_id = _xml_value(content, "tool-use-id") - tool = tool_uses.get(tool_id) - result = _xml_value(content, "result") - if tool and tool[0] == "Agent" and result: - assignment = _agent_assignment(tool[1]) - response = _agent_response(tool[1], result) - append("assistant", assignment) - append("assistant", response) - continue - human = _human_prompt_text(row) - if human: - append("user", human) - continue - - if not isinstance(content, list): - continue - for block in content: - if not isinstance(block, dict): - continue - block_type = block.get("type") - if role == "assistant" and block_type == "text": - append("assistant", str(block.get("text") or "")) - continue - if role != "user" or block_type != "tool_result": - continue - - tool_id = str(block.get("tool_use_id") or "") - tool = tool_uses.get(tool_id) - if not tool: - continue - name, tool_input = tool - result = _tool_result_text(block) - failed = bool(block.get("is_error")) - if name == "Agent" and not failed: - response = _agent_response(tool_input, result) - if response: - append("assistant", _agent_assignment(tool_input)) - append("assistant", response) - elif name == "AskUserQuestion" and result and not failed: - append("user", f"User answers to Claude's questions:\n{result}") - elif name == "ExitPlanMode" and not failed: - plan = redact(tool_input.get("plan", "")).strip() - if plan: - append("assistant", f"Approved implementation plan:\n{plan}") - - fallback = redact(fallback_assistant_message).strip() - if fallback: - labeled = f"Main Claude response:\n{fallback}" - for message in reversed(output): - if message["role"] == "assistant" and message["content"] == fallback: - message["content"] = labeled - break - else: - append("assistant", labeled) - elif label_final_response: - last_message = chain[-1].get("message") or {} - last_content = ( - last_message.get("content") if isinstance(last_message, dict) else None - ) - final_parts = [ - redact(block.get("text", "")).strip() - for block in (last_content if isinstance(last_content, list) else []) - if isinstance(block, dict) - and block.get("type") == "text" - and redact(block.get("text", "")).strip() - ] - for start in range(len(output) - len(final_parts), -1, -1): - candidate = output[start : start + len(final_parts)] - if final_parts and [item["content"] for item in candidate] == final_parts: - candidate[0]["content"] = ( - f"Main Claude response:\n{candidate[0]['content']}" - ) - break - return output, leaf_uuid, end_offset - - def _checkpoint_message(event: dict[str, Any]) -> str: kind = event.get("kind") payload = event.get("payload") or {} if kind == "user_prompt": - return bounded(payload.get("text", ""), MAX_PROMPT_CHARS) + return redact(payload.get("text", "")).strip() if kind == "assistant_stop": transcript_messages = payload.get("transcript_messages") or [] if isinstance(transcript_messages, list): @@ -784,9 +523,9 @@ def _checkpoint_message(event: dict[str, Any]) -> str: ) if text: return text - return bounded(payload.get("text", ""), MAX_ASSISTANT_CHARS) + return redact(payload.get("text", "")).strip() if kind == "sidekick_stop": - return bounded(payload.get("final_message", ""), MAX_ASSISTANT_CHARS) + return redact(payload.get("final_message", "")).strip() return "" @@ -998,8 +737,27 @@ class EvidenceStore: self.conn.commit() return int(cursor.lastrowid) + def record_assistant_response(self, repo: RepoContext, session_id: str, message: str) -> None: + """Ignore repeated response hooks until another prompt or a different answer arrives.""" + with self.conn: + # Serialize the check and insert across concurrent Stop and SessionEnd hooks. + self.conn.execute("BEGIN IMMEDIATE") + previous = self.conn.execute( + """SELECT kind, payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind IN ('user_prompt', 'assistant_stop') + ORDER BY id DESC LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if ( + previous is not None + and previous["kind"] == "assistant_stop" + and json.loads(previous["payload_json"]).get("text") == message + ): + return + self.record_event(repo, session_id, "assistant_stop", {"text": message}) + def repo_for_session(self, session_id: str, cwd: str | None) -> RepoContext: - """Keep one project scope for every hook in a Claude Code session.""" + """Keep one project scope for every hook in a coding-agent session.""" current = resolve_repo(cwd) if session_id == "unknown-session": return current @@ -1041,78 +799,77 @@ class EvidenceStore: def prepare_flush( self, repo: RepoContext, session_id: str, reason: str ) -> tuple[str, list[dict[str, Any]]] | None: - existing = self.conn.execute( - """SELECT * FROM flushes - WHERE repo_id = ? AND session_id = ? - AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') - ORDER BY created_at LIMIT 1""", - (repo.identity, session_id), - ).fetchone() - if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: - with self.conn: + with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + existing = self.conn.execute( + """SELECT * FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') + ORDER BY created_at LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: self.conn.execute( "UPDATE flushes SET status = 'gave-up', updated_at = ? WHERE packet_id = ?", (utc_now(), existing["packet_id"]), ) - telemetry.record( - "flush", - repo=repo, - session_id=session_id, - reason=reason, - status="gave-up", - success=False, - attempts=int(existing["attempts"] or 0), - ) - existing = None - if existing: - if reason != "periodic" and existing["reason"] == "periodic": - with self.conn: + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status="gave-up", + success=False, + attempts=int(existing["attempts"] or 0), + ) + existing = None + if existing: + if reason != "periodic" and existing["reason"] == "periodic": self.conn.execute( "UPDATE flushes SET reason = ?, updated_at = ? WHERE packet_id = ?", (reason, utc_now(), existing["packet_id"]), ) - existing_rows = self.conn.execute( - "SELECT * FROM events WHERE flush_id = ? ORDER BY id", - (existing["packet_id"],), + existing_rows = self.conn.execute( + "SELECT * FROM events WHERE flush_id = ? ORDER BY id", + (existing["packet_id"],), + ).fetchall() + if existing_rows: + return str(existing["packet_id"]), [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in existing_rows + ] + + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo.identity, session_id), ).fetchall() - if existing_rows: - return str(existing["packet_id"]), [ - { - "id": row["id"], - "created_at": row["created_at"], - "kind": row["kind"], - "payload": json.loads(row["payload_json"]), - } - for row in existing_rows - ] + if not rows: + return None - rows = self.conn.execute( - """SELECT * FROM events - WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL - ORDER BY id""", - (repo.identity, session_id), - ).fetchall() - if not rows: - return None + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + events = select_checkpoint_events(events, force=reason != "periodic") + if not events: + return None + event_start, event_end = events[0]["id"], events[-1]["id"] + packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" + packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] + now = utc_now() - events = [ - { - "id": row["id"], - "created_at": row["created_at"], - "kind": row["kind"], - "payload": json.loads(row["payload_json"]), - } - for row in rows - ] - events = select_checkpoint_events(events, force=reason != "periodic") - if not events: - return None - event_start, event_end = events[0]["id"], events[-1]["id"] - packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" - packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] - now = utc_now() - - with self.conn: self.conn.execute( """INSERT OR IGNORE INTO flushes (packet_id, repo_id, app_id, session_id, reason, event_start, @@ -1137,7 +894,7 @@ class EvidenceStore: f"WHERE id IN ({placeholders}) AND flush_id IS NULL", (packet_id, *event_ids), ) - return packet_id, events + return packet_id, events def checkpoint_due(self, repo_id: str, session_id: str) -> bool: if self.has_inflight_flush(repo_id, session_id): @@ -1318,9 +1075,19 @@ class EvidenceStore: agent_type: str, transcript_path: str, final_message: str, - ) -> None: + ) -> str: now = utc_now() - with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + try: + if not agent_id: + rows = self.conn.execute( + """SELECT agent_id FROM sidekick_runs + WHERE repo_id = ? AND session_id = ? AND agent_type = ? AND stopped_at IS NULL + LIMIT 2""", + (repo.identity, session_id, agent_type), + ).fetchall() + # Without a host ID, overlapping runs cannot be correlated reliably. + agent_id = rows[0]["agent_id"] if len(rows) == 1 else f"unknown-agent-{time.time_ns()}" self.conn.execute( """INSERT INTO sidekick_runs (repo_id, session_id, agent_id, agent_type, started_at, @@ -1338,9 +1105,14 @@ class EvidenceStore: now, now, bounded(transcript_path, 2000), - bounded(final_message, MAX_ASSISTANT_CHARS), + redact(final_message).strip(), ), ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + return agent_id def operation( self, @@ -1518,7 +1290,7 @@ def record_user_prompt( ) -> tuple[RepoContext, str, str, bool]: session_id = _session_id(hook_input) repo = store.repo_for_session(session_id, hook_input.get("cwd")) - prompt = bounded(hook_input.get("prompt", ""), MAX_PROMPT_CHARS) + prompt = redact(hook_input.get("prompt", "")).strip() is_first_prompt = not store.has_event(repo.identity, session_id, "user_prompt") store.record_event(repo, session_id, "user_prompt", {"text": prompt}) return repo, session_id, prompt, is_first_prompt @@ -1543,7 +1315,7 @@ def _tool_result_preview(response: Any) -> str: return bounded(response, MAX_RESULT_CHARS) -def tool_payload(hook_input: dict[str, Any], *, failed: bool = False) -> dict[str, Any]: +def tool_payload(hook_input: dict[str, Any], *, failed: bool | None = False) -> dict[str, Any]: name = str(hook_input.get("tool_name") or "unknown") tool_input = hook_input.get("tool_input") or {} if not isinstance(tool_input, dict): @@ -1597,7 +1369,7 @@ def tool_payload(hook_input: dict[str, Any], *, failed: bool = False) -> dict[st def record_tool( - store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool = False + store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool | None = False ) -> None: session_id = _session_id(hook_input) repo = store.repo_for_session(session_id, hook_input.get("cwd")) @@ -1612,51 +1384,8 @@ def record_tool( ) -def record_stop( - store: EvidenceStore, hook_input: dict[str, Any] -) -> tuple[RepoContext, str]: - session_id = _session_id(hook_input) - repo = store.repo_for_session(session_id, hook_input.get("cwd")) - message = bounded(hook_input.get("last_assistant_message", ""), MAX_ASSISTANT_CHARS) - previous_stop = store.latest_event_payload( - repo.identity, session_id, "assistant_stop" - ) - latest_prompt = store.latest_event_payload(repo.identity, session_id, "user_prompt") - transcript_path = str(hook_input.get("transcript_path") or "") - previous_offset = previous_stop.get("transcript_offset") - start_offset = ( - previous_offset - if isinstance(previous_offset, int) - and str(previous_stop.get("transcript_path") or "") == transcript_path - else 0 - ) - transcript_messages, leaf_uuid, end_offset = transcript_extraction_messages( - transcript_path, - session_id, - previous_leaf_uuid=str(previous_stop.get("transcript_leaf_uuid") or ""), - prompt_hint=str(latest_prompt.get("text") or ""), - fallback_assistant_message=message, - label_final_response=True, - start_offset=start_offset, - ) - if transcript_messages: - payload: dict[str, Any] = { - "text": message, - "transcript_messages": transcript_messages, - } - if leaf_uuid: - payload["transcript_leaf_uuid"] = leaf_uuid - if transcript_path: - payload["transcript_path"] = transcript_path - payload["transcript_offset"] = end_offset - store.record_event( - repo, session_id, "assistant_stop", payload - ) - return repo, session_id - - def record_sidekick_start( - store: EvidenceStore, hook_input: dict[str, Any] + store: EvidenceStore, hook_input: dict[str, Any], *, inject_context: bool = True ) -> str: """Record a native sidekick and reuse the main turn's retrieved memories.""" session_id = _session_id(hook_input) @@ -1666,6 +1395,8 @@ def record_sidekick_start( context = combine_context( format_context(store.injected_memories(session_id, repo.identity)) ) + if not inject_context: + context = "" first_start = store.start_sidekick( repo, session_id, agent_id, agent_type, len(context) ) @@ -1694,13 +1425,11 @@ def record_sidekick_start( def record_sidekick_stop(store: EvidenceStore, hook_input: dict[str, Any]) -> None: session_id = _session_id(hook_input) repo = store.repo_for_session(session_id, hook_input.get("cwd")) - agent_id = bounded(hook_input.get("agent_id", "unknown-agent"), 200) agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) - final_message = bounded( - hook_input.get("last_assistant_message", ""), MAX_ASSISTANT_CHARS - ) + agent_id = bounded(hook_input.get("agent_id", ""), 200) + final_message = redact(hook_input.get("last_assistant_message", "")).strip() transcript_path = bounded(hook_input.get("agent_transcript_path", ""), 2000) - store.stop_sidekick( + agent_id = store.stop_sidekick( repo, session_id, agent_id, @@ -1775,17 +1504,17 @@ def build_episode( task_outcome: str = "", ) -> tuple[str, dict[str, Any]]: prompts = [ - bounded(e["payload"].get("text", ""), MAX_PROMPT_CHARS) + redact(e["payload"].get("text", "")).strip() for e in events if e["kind"] == "user_prompt" and e["payload"].get("text") ] assistant_conclusions = [ - bounded(e["payload"].get("text", ""), MAX_ASSISTANT_CHARS) + redact(e["payload"].get("text", "")).strip() for e in events if e["kind"] == "assistant_stop" and e["payload"].get("text") ] sidekick_outcomes = [ - bounded(e["payload"].get("final_message", ""), MAX_ASSISTANT_CHARS) + redact(e["payload"].get("final_message", "")).strip() for e in events if e["kind"] == "sidekick_stop" and e["payload"].get("final_message") ] @@ -1812,7 +1541,7 @@ def build_episode( { "command": t.get("command", ""), "kind": t.get("command_kind", "shell"), - "status": "failed" if t.get("failed") else "succeeded", + "status": "unknown" if t.get("failed", False) is None else "failed" if t.get("failed") else "succeeded", "result": t.get("result_preview", ""), } for t in tools @@ -1820,10 +1549,7 @@ def build_episode( ] task = bounded(canonical_task or (prompts[0] if prompts else ""), 4000) - conclusion = bounded( - assistant_conclusions[-1] if assistant_conclusions else "", - MAX_ASSISTANT_CHARS, - ) + conclusion = redact(assistant_conclusions[-1] if assistant_conclusions else "").strip() outcome = bounded(task_outcome, 2000) extraction_messages: list[dict[str, str]] = [] @@ -1835,16 +1561,14 @@ def build_episode( pending_user_messages.append( { "role": "user", - "content": bounded( - event["payload"].get("text", ""), MAX_PROMPT_CHARS - ), + "content": redact(event["payload"].get("text", "")).strip(), } ) elif event["kind"] == "assistant_stop": transcript_messages = event["payload"].get("transcript_messages") or [] if isinstance(transcript_messages, list) and transcript_messages: transcript_users = { - str(message.get("content") or "").strip() + redact(message.get("content") or "").strip() for message in transcript_messages if isinstance(message, dict) and message.get("role") == "user" } @@ -1856,7 +1580,7 @@ def build_episode( extraction_messages.extend( { "role": str(message.get("role") or ""), - "content": str(message.get("content") or ""), + "content": redact(message.get("content") or "").strip(), } for message in transcript_messages if isinstance(message, dict) @@ -1869,10 +1593,7 @@ def build_episode( extraction_messages.append( { "role": "assistant", - "content": bounded( - event["payload"].get("text", ""), - MAX_ASSISTANT_CHARS, - ), + "content": redact(event["payload"].get("text", "")).strip(), } ) pending_user_messages = [] @@ -1972,7 +1693,7 @@ def build_extraction_messages(structured: dict[str, Any]) -> list[dict[str, str] """Build the session messages sent to Mem0 for memory extraction.""" evidence = build_semantic_evidence(structured) messages = [ - {"role": message["role"], "content": message["content"]} + {"role": message["role"], "content": redact(message["content"]).strip()} for message in structured.get("extraction_messages", []) if message.get("role") in {"user", "assistant"} and message.get("content") ] @@ -2014,7 +1735,7 @@ def extraction_message_batches( *, max_tokens: int = MAX_EXTRACTION_INPUT_TOKENS, ) -> list[list[dict[str, str]]]: - """Split large extraction input without cutting messages or agent pairs.""" + """Keep exchanges together when possible; split oversized messages to enforce the request budget.""" if not messages or _message_tokens(messages) <= max_tokens: return [messages] @@ -2047,9 +1768,29 @@ def extraction_message_batches( units.append([message]) index += 1 + bounded_units: list[list[dict[str, str]]] = [] + for unit in units: + if _message_tokens(unit) <= max_tokens: + bounded_units.append(unit) + continue + for message in unit: + remaining = message["content"] + while remaining: + low, high = 0, len(remaining) + while low < high: + middle = (low + high + 1) // 2 + if _message_tokens([{**message, "content": remaining[:middle]}]) <= max_tokens: + low = middle + else: + high = middle - 1 + if low == 0: + raise ValueError("Extraction token budget cannot fit a message") + bounded_units.append([{**message, "content": remaining[:low]}]) + remaining = remaining[low:] + batches: list[list[dict[str, str]]] = [] batch: list[dict[str, str]] = [] - for unit in units: + for unit in bounded_units: candidate = [*batch, *unit] if batch and _message_tokens(candidate) > max_tokens: batches.append(batch) @@ -2152,6 +1893,11 @@ def _wait_for_event(api_url: str, key: str, event_id: str) -> tuple[str, int, in key, min(10, poll_seconds + 5), ) + except urllib.error.HTTPError as exc: + if exc.code not in {408, 429} and exc.code < 500: + raise + time.sleep(poll_seconds) + continue except (urllib.error.URLError, TimeoutError, OSError): # The extraction job is durable server-side. A transient polling # failure must not discard a job that may still complete normally. @@ -2217,7 +1963,7 @@ def flush_session( task_outcome=bounded(hook_input.get("task_outcome", ""), 2000), ) - metadata = {"source": "claude_code_plugin"} + metadata = {"source": _harness_source_tag} if repo.branch and repo.branch not in {"detached", "unknown"}: metadata["branch"] = repo.branch if repo.head_sha: @@ -2684,7 +2430,7 @@ def format_context( def format_search_result(result: MemorySearchResult) -> str: - """Return only the text Claude needs from an explicit memory search.""" + """Return only the text the coding agent needs from an explicit memory search.""" if not result.succeeded: return "Memory search failed." if result.memories: @@ -2731,7 +2477,11 @@ def _scoped_memory_ids( app_id_prefix=prefix, ) if include_project: - _collect_memory_ids(api_url, key, {"agent_id": repo.project_id}, ids, seen) + for project_id in _shared_project_ids(repo): + _collect_memory_ids( + api_url, key, {"agent_id": project_id}, ids, seen, + app_id_prefix=prefix, + ) return ids diff --git a/integrations/claude-code-plugin/core/telemetry.py b/integrations/claude-code-plugin/core/telemetry.py index 2b704a339..249595475 100644 --- a/integrations/claude-code-plugin/core/telemetry.py +++ b/integrations/claude-code-plugin/core/telemetry.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Anonymous usage telemetry for the Mem0 Claude Code plugin. +"""Anonymous usage telemetry for Mem0 agent plugins. Hooks run on a 3-6 second budget and fire on every tool call, so recording never touches the network: `record` appends one JSON line to a local spool and returns. @@ -29,6 +29,38 @@ from typing import Any import memory_core +_harness: str = "generic" +_source_tag: str = "MEM0_PLUGIN" +_PRIVATE_KEYS = { + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +} + + +def init(harness: str = "generic", source_tag: str = "") -> None: + global _harness, _source_tag + _harness = harness + _source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN" + POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/" @@ -54,6 +86,23 @@ def _digest(value: str, length: int = 16) -> str: return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length] +def _safe_value(value: Any) -> Any: + if isinstance(value, str): + return memory_core.redact(value) + if isinstance(value, dict): + return { + key: _safe_value(item) + for key, item in value.items() + if "".join(character for character in str(key).lower() if character.isalnum()) + not in _PRIVATE_KEYS + } + if isinstance(value, (list, tuple)): + return [_safe_value(item) for item in value] + if value is None or isinstance(value, (bool, int, float)): + return value + return memory_core.redact(value) + + def _spool_path() -> Path: return memory_core.data_dir() / "telemetry.jsonl" @@ -118,8 +167,9 @@ def record( return except OSError: pass + properties = _safe_value(properties) properties.update( - harness="claude-code", + harness=_harness, plugin_version=memory_core.PLUGIN_VERSION, os=sys.platform, python_version=platform.python_version(), @@ -316,7 +366,7 @@ def flush() -> int: "distinct_id": distinct_id, "timestamp": event.get("timestamp"), "properties": { - "source": "CLAUDE_CODE_PLUGIN", + "source": _source_tag, "language": "python", "$process_person_profile": False, "$lib": "posthog-python", diff --git a/integrations/claude-code-plugin/plugin-build.json b/integrations/claude-code-plugin/plugin-build.json new file mode 100644 index 000000000..68d306fd6 --- /dev/null +++ b/integrations/claude-code-plugin/plugin-build.json @@ -0,0 +1,17 @@ +{ + "id": "mem0", + "version": "0.3.1", + "homepage": "https://docs.mem0.ai/integrations/claude-code", + "native": { + "pluginRoot": "${CLAUDE_PLUGIN_ROOT}", + "pluginData": "${CLAUDE_PLUGIN_DATA}", + "files": { + ".claude-plugin/plugin.json": ".claude-plugin/plugin.json", + ".mcp.json": ".mcp.json", + "hooks/hooks.json": "hooks/hooks.json", + "adapters/claude/hook.py": "adapters/claude/hook.py", + "adapters/claude/transcript.py": "adapters/claude/transcript.py", + "agents/sidekick.md": "agents/sidekick.md" + } + } +} diff --git a/integrations/claude-code-plugin/skills/forget/SKILL.md b/integrations/claude-code-plugin/skills/forget/SKILL.md index c61219cc6..7cf25a38e 100644 --- a/integrations/claude-code-plugin/skills/forget/SKILL.md +++ b/integrations/claude-code-plugin/skills/forget/SKILL.md @@ -14,7 +14,7 @@ so it stays unless the user explicitly asks to delete that too. After the user confirms, run: ```bash -python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" forget --remote --yes +python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" --harness "claude-code" --plugin-data-dir "${CLAUDE_PLUGIN_DATA}" forget --remote --yes ``` If the user also asked to delete the repository's shared project memory, add diff --git a/integrations/claude-code-plugin/skills/pause/SKILL.md b/integrations/claude-code-plugin/skills/pause/SKILL.md index 6492ccc43..5aef04f39 100644 --- a/integrations/claude-code-plugin/skills/pause/SKILL.md +++ b/integrations/claude-code-plugin/skills/pause/SKILL.md @@ -11,7 +11,7 @@ anonymous telemetry ping still fires at session start unless `MEM0_TELEMETRY=false`): ```bash -python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" pause +python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" --harness "claude-code" --plugin-data-dir "${CLAUDE_PLUGIN_DATA}" pause ``` Confirm the new state back to the user, and remind them that already-created diff --git a/integrations/claude-code-plugin/skills/resume/SKILL.md b/integrations/claude-code-plugin/skills/resume/SKILL.md index f4877fc9f..61082fb42 100644 --- a/integrations/claude-code-plugin/skills/resume/SKILL.md +++ b/integrations/claude-code-plugin/skills/resume/SKILL.md @@ -11,7 +11,7 @@ Resume memory capture for this machine. Run: ```bash -python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" resume +python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" --harness "claude-code" --plugin-data-dir "${CLAUDE_PLUGIN_DATA}" resume ``` Confirm to the user that capture is active again. New sessions record evidence and diff --git a/integrations/claude-code-plugin/skills/search/SKILL.md b/integrations/claude-code-plugin/skills/search/SKILL.md index ad5c958bb..198a1f023 100644 --- a/integrations/claude-code-plugin/skills/search/SKILL.md +++ b/integrations/claude-code-plugin/skills/search/SKILL.md @@ -20,7 +20,9 @@ preferences. Pass `scope` when the question needs something else: `dir` to narrow the shared memory to the directory you are working in (a package inside a -monorepo), `mine` for your own preferences alone. Pass `run_id` with a Claude -Code session ID to look at what one earlier session recorded, for example to -pick up where a compacted or closed session left off. Return the tool's result -directly. +monorepo), `mine` for your own preferences alone. + +Pass `run_id` with any scope to retrieve memories saved in a specific coding-agent +session. Omit `run_id` to search across sessions. It filters the memories returned; +it does not identify the session making the search request. Use a known session ID, +never invent one. Return the tool's result directly. diff --git a/integrations/claude-code-plugin/skills/status/SKILL.md b/integrations/claude-code-plugin/skills/status/SKILL.md index ef572549c..ed5e49b3f 100644 --- a/integrations/claude-code-plugin/skills/status/SKILL.md +++ b/integrations/claude-code-plugin/skills/status/SKILL.md @@ -9,8 +9,8 @@ disable-model-invocation: false Run both commands and report the combined result in plain language: ```bash -python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" status --json -python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" doctor +python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" --harness "claude-code" --plugin-data-dir "${CLAUDE_PLUGIN_DATA}" status --json +python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" --harness "claude-code" --plugin-data-dir "${CLAUDE_PLUGIN_DATA}" doctor ``` Summarize, using only fields the JSON actually reports: whether capture is diff --git a/integrations/claude-code-plugin/tests/conftest.py b/integrations/claude-code-plugin/tests/conftest.py index 85245a5ee..a5b795951 100644 --- a/integrations/claude-code-plugin/tests/conftest.py +++ b/integrations/claude-code-plugin/tests/conftest.py @@ -1,7 +1,20 @@ -"""Keep the test suite from sending usage telemetry to the live PostHog project.""" +"""Configure harness identity and disable live telemetry for the test suite.""" from __future__ import annotations import os +import sys +from pathlib import Path os.environ["MEM0_TELEMETRY"] = "false" + +HOST_ROOT = Path(__file__).resolve().parents[1] +_core = HOST_ROOT / "core" +sys.path.insert(0, str(_core)) + +import hook_runner # noqa: E402,F401 +from memory_core import configure_harness # noqa: E402 +import telemetry # noqa: E402 + +configure_harness("claude-code", data_dir_name="claude-code-plugin", source_tag="claude_code_plugin") +telemetry.init(harness="claude-code", source_tag="CLAUDE_CODE_PLUGIN") diff --git a/integrations/claude-code-plugin/tests/integration/test_live_scoping.py b/integrations/claude-code-plugin/tests/integration/test_live_scoping.py index da223f5c8..90542868f 100644 --- a/integrations/claude-code-plugin/tests/integration/test_live_scoping.py +++ b/integrations/claude-code-plugin/tests/integration/test_live_scoping.py @@ -13,10 +13,11 @@ from pathlib import Path import pytest PLUGIN_ROOT = Path(__file__).resolve().parents[2] -sys.path.insert(0, str(PLUGIN_ROOT / "core")) -sys.path.insert(0, str(PLUGIN_ROOT / "adapters" / "claude")) +_core_dir = PLUGIN_ROOT / "core" +sys.path.insert(0, str(_core_dir)) +sys.path.insert(0, str(PLUGIN_ROOT)) -import hook # noqa: E402 +import hook_runner # noqa: E402 import memory_core # noqa: E402 pytestmark = pytest.mark.skipif( @@ -177,15 +178,17 @@ def test_shared_project_memory_reaches_a_teammate_who_never_wrote(ns): def test_repo_scope_spans_every_subdirectory(ns): found = ns.search("erin", ns.root, "how many times are Stripe webhooks retried", scope="repo") assert "stripe" in _text(found) - app_ids = {m.get("app_id") for m in found} - assert any(app.endswith("/services/billing") for app in app_ids) + assert any("services/billing" in (memory.get("metadata") or {}).get("dirs", []) for memory in found) def test_dir_scope_narrows_shared_memory_to_the_directory(ns): billing = ns.root / "services" / "billing" found = ns.search("erin", billing, "how do I run the tests here", scope="dir") assert "billing-test" in _text(found) - assert {m.get("app_id") for m in found if m.get("agent_id")} == {memory_core.directory_app_id(memory_core.resolve_repo(str(billing)))} + shared = [memory for memory in found if memory.get("agent_id")] + assert shared + assert {memory.get("app_id") for memory in shared} == {memory_core.resolve_repo(str(billing)).app_id} + assert all("services/billing" in (memory.get("metadata") or {}).get("dirs", []) for memory in shared) web = ns.root / "apps" / "web" elsewhere = ns.search("erin", web, "Stripe webhook retries exponential backoff", scope="dir", tries=1) @@ -232,7 +235,7 @@ def test_a_pending_packet_is_recovered_and_delivered_by_the_worker(ns): stale.write_text(json.dumps({"hook_input": {"session_id": sid, "cwd": str(ns.root)}, "reason": "session-end"})) os.utime(stale, (time.time() - 3600, time.time() - 3600)) - assert hook.recover_pending_handoffs() == 1 + assert hook_runner.recover_pending_handoffs() == 1 deadline = time.time() + 240 while time.time() < deadline and list(pending.iterdir()): time.sleep(3) diff --git a/integrations/claude-code-plugin/tests/test_claude_build.py b/integrations/claude-code-plugin/tests/test_claude_build.py new file mode 100644 index 000000000..46ccfb206 --- /dev/null +++ b/integrations/claude-code-plugin/tests/test_claude_build.py @@ -0,0 +1,35 @@ +from __future__ import annotations + +import sys +from pathlib import Path + +HOST = Path(__file__).resolve().parents[1] +CORE_ROOT = HOST.parent / "agent-plugin-core" +sys.path.insert(0, str(CORE_ROOT)) + +from build.build import SHARED_SKILLS, build, render_template # noqa: E402 + + +def test_native_claude_bundle_preserves_working_contract(tmp_path: Path) -> None: + root = build("claude-code", "native", tmp_path / "claude-code") + + assert (root / ".claude-plugin" / "plugin.json").read_bytes() == ( + HOST / ".claude-plugin" / "plugin.json" + ).read_bytes() + assert (root / ".mcp.json").read_bytes() == (HOST / ".mcp.json").read_bytes() + assert (root / "hooks" / "hooks.json").read_bytes() == (HOST / "hooks" / "hooks.json").read_bytes() + assert (root / "agents" / "sidekick.md").read_bytes() == (HOST / "agents" / "sidekick.md").read_bytes() + values = { + "PLUGIN_ROOT": "${CLAUDE_PLUGIN_ROOT}", + "PLUGIN_DATA": "${PLUGIN_DATA}", + "PLUGIN_DATA_ARG": '--plugin-data-dir "${CLAUDE_PLUGIN_DATA}"', + "COMMAND_PREFIX": "mem0", + "HARNESS_ID": "claude-code", + "HARNESS_NAME": "Claude Code", + } + for skill in SHARED_SKILLS.glob("*/SKILL.md.tmpl"): + rendered = render_template(skill.read_text(encoding="utf-8"), values) + assert (root / "skills" / skill.parent.name / "SKILL.md").read_text(encoding="utf-8") == rendered + assert (root / "adapters" / "claude" / "hook.py").is_file() + assert (root / "core" / "mcp_server.py").is_file() + assert not any(path.is_symlink() for path in root.rglob("*")) diff --git a/integrations/claude-code-plugin/tests/test_memory_core.py b/integrations/claude-code-plugin/tests/test_memory_core.py index 23bf72f8a..6c91b7299 100644 --- a/integrations/claude-code-plugin/tests/test_memory_core.py +++ b/integrations/claude-code-plugin/tests/test_memory_core.py @@ -11,16 +11,18 @@ from unittest.mock import patch import pytest - -PLUGIN_ROOT = Path(__file__).resolve().parents[1] -CORE = PLUGIN_ROOT / "core" -ADAPTER = PLUGIN_ROOT / "adapters" / "claude" +HOST_ROOT = Path(__file__).resolve().parents[1] +REPOSITORY_ROOT = HOST_ROOT.parents[1] +PLUGIN_ROOT = HOST_ROOT +CORE = HOST_ROOT / "core" +ADAPTER = HOST_ROOT / "adapters" / "claude" / "hook.py" sys.path.insert(0, str(CORE)) -sys.path.insert(0, str(ADAPTER)) +sys.path.insert(0, str(ADAPTER.parent)) -import memory_core # noqa: E402 -import memory_cli # noqa: E402 import mcp_server # noqa: E402 +import memory_cli # noqa: E402 +import memory_core # noqa: E402 +import transcript as transcript_mod # noqa: E402 @pytest.fixture @@ -54,6 +56,14 @@ def repo() -> memory_core.RepoContext: ) +def test_portable_plugin_uses_standard_data_directory(tmp_path, monkeypatch): + monkeypatch.delenv("MEM0_CODE_DATA_DIR", raising=False) + monkeypatch.delenv("CLAUDE_PLUGIN_DATA", raising=False) + monkeypatch.setenv("PLUGIN_DATA", str(tmp_path)) + + assert memory_core.data_dir() == tmp_path + + def _write_transcript(path: Path, session_id: str, entries: list[dict]) -> None: parent = None rows = [] @@ -235,7 +245,7 @@ def test_transcript_extraction_keeps_meaningful_messages_and_excludes_raw_tools( ], ) - messages, leaf, _ = memory_core.transcript_extraction_messages( + messages, leaf, _ = transcript_mod.transcript_extraction_messages( str(transcript), "s1" ) serialized = json.dumps(messages) @@ -335,7 +345,7 @@ def test_transcript_extraction_pairs_background_agent_notification(tmp_path): ], ) - messages, _, _ = memory_core.transcript_extraction_messages( + messages, _, _ = transcript_mod.transcript_extraction_messages( str(transcript), "s1", previous_leaf_uuid="launch-result" ) serialized = json.dumps(messages) @@ -388,7 +398,7 @@ def test_transcript_extraction_omits_claude_ui_messages_but_keeps_attachments( ], ) - messages, _, _ = memory_core.transcript_extraction_messages( + messages, _, _ = transcript_mod.transcript_extraction_messages( str(transcript), "s1", prompt_hint="The attached issue says repository scope is wrong." ) serialized = json.dumps(messages) @@ -487,7 +497,7 @@ def test_transcript_extraction_uses_active_branch_and_omits_rejected_plan(tmp_pa "".join(json.dumps(row) + "\n" for row in rows), encoding="utf-8" ) - messages, leaf, _ = memory_core.transcript_extraction_messages( + messages, leaf, _ = transcript_mod.transcript_extraction_messages( str(transcript), "s1" ) serialized = json.dumps(messages) @@ -525,7 +535,7 @@ def test_record_stop_processes_only_new_transcript_entries( store, {"session_id": "s1", "cwd": "/tmp/repo", "prompt": "Inspect the parser."}, ) - memory_core.record_stop( + transcript_mod.record_stop( store, { "session_id": "s1", @@ -562,7 +572,7 @@ def test_record_stop_processes_only_new_transcript_entries( "prompt": "Where is that implemented?", }, ) - memory_core.record_stop( + transcript_mod.record_stop( store, { "session_id": "s1", @@ -591,7 +601,7 @@ def test_record_stop_processes_only_new_transcript_entries( ] -def test_extraction_messages_keep_long_content(): +def test_extraction_messages_preserve_long_content(): long_response = "repository detail " * 4000 structured = { "extraction_messages": [ @@ -605,10 +615,47 @@ def test_extraction_messages_keep_long_content(): assert messages == [ {"role": "user", "content": "Explain the repository."}, - {"role": "assistant", "content": long_response}, + {"role": "assistant", "content": long_response.strip()}, ] +@pytest.mark.parametrize("capture", ["hook", "transcript"]) +def test_long_conversation_reaches_extraction_without_truncation(isolated_env, monkeypatch, capture): + import hook_runner + + monkeypatch.setenv("MEM0_API_KEY", "test-key") + monkeypatch.setattr(memory_core, "resolve_repo", lambda cwd: repo()) + prompt = "Repository question. " * 2000 + "Final user requirement. api_key=hidden-user-secret" + answer = "Repository answer. " * 4000 + "Final implementation detail. password=hidden-agent-secret" + payload = {"session_id": "s1", "cwd": "/tmp/repo", "prompt": prompt, "last_assistant_message": answer} + store = memory_core.EvidenceStore() + memory_core.record_user_prompt(store, payload) + if capture == "transcript": + transcript = isolated_env / "long-session.jsonl" + _write_transcript(transcript, "s1", [ + {"type": "user", "message": {"role": "user", "content": prompt}}, + {"type": "assistant", "message": {"role": "assistant", "content": answer}}, + ]) + transcript_mod.record_stop(store, {**payload, "transcript_path": str(transcript)}) + else: + hook_runner.default_record_stop(store, payload) + + with ( + patch.object(memory_core, "_request_json", return_value=({"event_id": "extraction"}, 200, 20)) as request, + patch.object(memory_core, "_wait_for_event", return_value=("SUCCEEDED", 20, 1)), + ): + assert memory_core.flush_session(store, payload, "session-end")["status"] == "semantic-succeeded" + + batches = [call.args[2]["messages"] for call in request.call_args_list] + assert len(batches) > 1 + assert all(memory_core._message_tokens(batch) <= memory_core.MAX_EXTRACTION_INPUT_TOKENS for batch in batches) + for role, text in (("user", prompt), ("assistant", answer)): + actual = "".join(message["content"] for batch in batches for message in batch if message["role"] == role) + prefix = "Main Claude response:\n" if capture == "transcript" and role == "assistant" else "" + assert actual == prefix + memory_core.redact(text) + store.close() + + def test_upgrade_removes_legacy_snapshot_tables(isolated_env): database = Path(os.environ["MEM0_CODE_DATA_DIR"]) / "evidence.sqlite3" database.parent.mkdir(parents=True, exist_ok=True) @@ -863,6 +910,20 @@ def test_common_tokens_and_private_keys_are_redacted(): assert "private-material" not in value +def test_json_shaped_secrets_are_redacted(): + value = memory_core.redact('{"apiKey": "sk-secret-12345", "name": "test"}') + assert "sk-secret-12345" not in value + assert "[REDACTED]" in value + assert '"name": "test"' in value + + value = memory_core.redact('{"password": "hunter2", "count": 1}') + assert "hunter2" not in value + + value = memory_core.redact('{"secret_access_key": "ABCDEFGHIJ1234567890", "region": "us-east-1"}') + assert "ABCDEFGHIJ1234567890" not in value + assert '"region": "us-east-1"' in value + + def test_remote_identity_removes_embedded_credentials(): normalized = memory_core._normalize_remote( "https://secret-token@github.com/example/repo.git" @@ -896,6 +957,49 @@ def test_repo_scope_matches_the_previous_plugin_mapping( assert resolved.app_id == "customer-platform" +def test_different_git_hosts_produce_different_project_ids(): + github = memory_core._project_id("/tmp/repo", "https://github.com/acme/api", "acme-api") + gitlab = memory_core._project_id("/tmp/repo", "https://gitlab.internal/acme/api", "acme-api") + + assert github != gitlab + assert github.startswith("acme-api-") + assert gitlab.startswith("acme-api-") + + +@pytest.mark.parametrize("scope", ["repo", "dir", "mine"]) +def test_upgraded_repo_search_retrieves_legacy_and_current_shared_memories(isolated_env, monkeypatch, scope): + from dataclasses import replace + + monkeypatch.setenv("MEM0_API_KEY", "test-key") + project = replace(repo(), project_id=memory_core._project_id(repo().root, repo().identity, repo().app_id)) + records = [ + {"id": "legacy", "agent_id": project.app_id, "app_id": project.app_id}, + {"id": "current", "agent_id": project.project_id, "app_id": project.app_id}, + {"id": "personal", "user_id": "test-user", "app_id": project.app_id}, + {"id": "teammate", "user_id": "someone-else", "app_id": project.app_id}, + {"id": "other-app", "agent_id": project.app_id, "app_id": "another-repo"}, + {"id": "other-host", "agent_id": "code-example-other-host", "app_id": project.app_id}, + ] + + def matches(record, filters): + for key, value in filters.items(): + if key == "AND": + return all(matches(record, child) for child in value) + if key == "OR": + return any(matches(record, child) for child in value) + if record.get(key) != value: + return False + return True + + def search(url, key, payload, timeout): + return {"results": [record for record in records if matches(record, payload["filters"])]}, 0, 0 + + monkeypatch.setattr(memory_core, "_request_json_with_network_retry", search) + result = memory_core.search_memories(None, project, None, "prior work", scope=scope) + expected = {"personal"} if scope == "mine" else {"legacy", "current", "personal"} + assert {memory["id"] for memory in result.memories} == expected + + def test_non_git_session_keeps_starting_project_scope_after_nested_commands( isolated_env, ): @@ -1061,32 +1165,32 @@ def test_forced_checkpoint_sends_an_incomplete_remainder(isolated_env): def test_stop_schedules_one_background_checkpoint_when_block_is_ready( isolated_env, ): - import hook + import hook_runner store = memory_core.EvidenceStore() for number in range(1, 5): _record_exchange(store, number) with ( - patch.object(hook, "api_key", return_value="m0-test-key"), - patch.object(hook, "hand_off_flush") as handoff, + patch.object(hook_runner, "api_key", return_value="m0-test-key"), + patch.object(hook_runner, "hand_off_flush") as handoff, ): assert ( - hook.schedule_periodic_checkpoint( + hook_runner.schedule_periodic_checkpoint( store, {"session_id": "s1", "cwd": "/tmp/repo"}, repo(), "s1" ) is False ) _record_exchange(store, 5) assert ( - hook.schedule_periodic_checkpoint( + hook_runner.schedule_periodic_checkpoint( store, {"session_id": "s1", "cwd": "/tmp/repo"}, repo(), "s1" ) is True ) _record_exchange(store, 6) assert ( - hook.schedule_periodic_checkpoint( + hook_runner.schedule_periodic_checkpoint( store, {"session_id": "s1", "cwd": "/tmp/repo"}, repo(), "s1" ) is False @@ -1100,64 +1204,64 @@ def test_stop_schedules_one_background_checkpoint_when_block_is_ready( def test_stop_schedules_idle_flush_when_periodic_not_due(isolated_env): - import hook + import hook_runner store = memory_core.EvidenceStore() _record_exchange(store, 1) hook_input = {"session_id": "s1", "cwd": "/tmp/repo"} with ( - patch.object(hook, "api_key", return_value="m0-test-key"), - patch.object(hook, "_launch_handoff", return_value=True) as launch, + patch.object(hook_runner, "api_key", return_value="m0-test-key"), + patch.object(hook_runner, "_launch_handoff", return_value=True) as launch, ): - assert not hook.schedule_periodic_checkpoint(store, hook_input, repo(), "s1") - assert hook.schedule_idle_flush(store, hook_input, repo(), "s1") + assert not hook_runner.schedule_periodic_checkpoint(store, hook_input, repo(), "s1") + assert hook_runner.schedule_idle_flush(store, hook_input, repo(), "s1") payload = json.loads(launch.call_args.args[0].read_text(encoding="utf-8")) assert payload["reason"] == "idle" - assert payload["delay_seconds"] == hook.DEFAULT_IDLE_FLUSH_SECONDS + assert payload["delay_seconds"] == hook_runner.DEFAULT_IDLE_FLUSH_SECONDS store.close() def test_idle_flush_skipped_when_inflight_flush_exists(isolated_env): - import hook + import hook_runner store = memory_core.EvidenceStore() for number in range(1, 6): _record_exchange(store, number) hook_input = {"session_id": "s1", "cwd": "/tmp/repo"} - with patch.object(hook, "api_key", return_value="m0-test-key"): + with patch.object(hook_runner, "api_key", return_value="m0-test-key"): store.prepare_flush(repo(), "s1", "periodic") - assert not hook.schedule_idle_flush(store, hook_input, repo(), "s1") + assert not hook_runner.schedule_idle_flush(store, hook_input, repo(), "s1") store.close() def test_idle_flush_skipped_when_no_unflushed_events(isolated_env): - import hook + import hook_runner store = memory_core.EvidenceStore() hook_input = {"session_id": "s1", "cwd": "/tmp/repo"} - with patch.object(hook, "api_key", return_value="m0-test-key"): - assert not hook.schedule_idle_flush(store, hook_input, repo(), "s1") + with patch.object(hook_runner, "api_key", return_value="m0-test-key"): + assert not hook_runner.schedule_idle_flush(store, hook_input, repo(), "s1") store.close() def test_idle_flush_skipped_when_disabled(isolated_env, monkeypatch): - import hook + import hook_runner store = memory_core.EvidenceStore() _record_exchange(store, 1) monkeypatch.setenv("MEM0_CODE_IDLE_FLUSH_SECONDS", "0") hook_input = {"session_id": "s1", "cwd": "/tmp/repo"} - with patch.object(hook, "api_key", return_value="m0-test-key"): - assert not hook.schedule_idle_flush(store, hook_input, repo(), "s1") + with patch.object(hook_runner, "api_key", return_value="m0-test-key"): + assert not hook_runner.schedule_idle_flush(store, hook_input, repo(), "s1") store.close() def test_stop_hook_falls_through_to_idle_flush(isolated_env, monkeypatch): - import hook + import hook_runner store = memory_core.EvidenceStore() _record_exchange(store, 1) @@ -1166,7 +1270,7 @@ def test_stop_hook_falls_through_to_idle_flush(isolated_env, monkeypatch): periodic_called = [] idle_called = [] - original_periodic = hook.schedule_periodic_checkpoint + original_periodic = hook_runner.schedule_periodic_checkpoint def mock_periodic(*a, **kw): result = original_periodic(*a, **kw) @@ -1178,14 +1282,17 @@ def test_stop_hook_falls_through_to_idle_flush(isolated_env, monkeypatch): return True with ( - patch.object(hook, "api_key", return_value="m0-test-key"), - patch.object(hook, "schedule_periodic_checkpoint", side_effect=mock_periodic), - patch.object(hook, "schedule_idle_flush", side_effect=mock_idle), - patch.object(hook, "record_stop", return_value=(repo(), "s1")), + patch.object(hook_runner, "api_key", return_value="m0-test-key"), + patch.object(hook_runner, "schedule_periodic_checkpoint", side_effect=mock_periodic), + patch.object(hook_runner, "schedule_idle_flush", side_effect=mock_idle), ): monkeypatch.setattr("sys.stdin", __import__("io").StringIO(json.dumps(hook_input))) monkeypatch.setattr("sys.argv", ["hook.py", "stop"]) - hook.main() + hook_runner.run( + record_stop_fn=lambda s, h: (repo(), "s1"), + data_dir_env="MEM0_CODE_DATA_DIR", + automatic_flush_reasons={"session-end", "pre-compact"}, + ) assert periodic_called == [False] assert idle_called == [True] @@ -1228,6 +1335,136 @@ def test_flush_worker_sleeps_for_delay_seconds(isolated_env, monkeypatch): assert "delay_seconds" not in rewritten_content +def test_session_end_worker_does_not_repeat_captured_final_response(isolated_env, monkeypatch): + import flush_worker + + monkeypatch.setenv("MEM0_API_KEY", "test-key") + monkeypatch.setattr(memory_core, "resolve_repo", lambda cwd: repo()) + store = memory_core.EvidenceStore() + for index in range(5): + store.record_event(repo(), "s1", "user_prompt", {"text": f"Earlier prompt {index}"}) + store.record_event(repo(), "s1", "assistant_stop", {"text": f"Earlier answer {index}"}) + first_packet, _ = store.prepare_flush(repo(), "s1", "periodic") + store.update_flush(first_packet, status="semantic-queued", semantic_event_id="earlier-extraction") + + prompt = "How does the serializer handle dates?" + answer = "The serializer preserves timezone-naive dates." + transcript = isolated_env / "session.jsonl" + _write_transcript(transcript, "s1", [ + {"type": "user", "message": {"role": "user", "content": prompt}}, + {"type": "assistant", "message": {"role": "assistant", "content": answer}}, + ]) + hook_input = { + "session_id": "s1", "cwd": "/tmp/repo", + "transcript_path": str(transcript), "last_assistant_message": answer, + } + memory_core.record_user_prompt(store, {**hook_input, "prompt": prompt}) + transcript_mod.record_stop(store, hook_input) + # SessionEnd captures again before handing off, using the same transcript cursor. + transcript_mod.record_stop(store, hook_input) + handoff = isolated_env / "session-end.running" + handoff.write_text(json.dumps({ + "hook_input": hook_input, "reason": "session-end", "wait_for_inflight": True, + })) + monkeypatch.setattr(sys, "argv", ["flush_worker.py", str(handoff)]) + monkeypatch.setenv("MEM0_CODE_HANDOFF_PATH", str(handoff)) + monkeypatch.setattr(flush_worker.time, "sleep", lambda seconds: store.update_flush( + first_packet, status="semantic-succeeded", + )) + with ( + patch.object(memory_core, "_request_json", return_value=({"event_id": "final-extraction"}, 200, 20)) as request, + patch.object(memory_core, "_wait_for_event", return_value=("SUCCEEDED", 20, 1)), + ): + assert flush_worker.main() == 0 + + assert request.call_count == 1 + messages = request.call_args.args[2]["messages"] + assert sum(prompt in message["content"] for message in messages) == 1 + assert sum(answer in message["content"] for message in messages) == 1 + assert not handoff.exists() + store.close() + + +def test_default_stop_records_each_exchange_once_across_repeated_hooks(isolated_env, monkeypatch): + import hook_runner + + monkeypatch.setattr(memory_core, "resolve_repo", lambda cwd: repo()) + store = memory_core.EvidenceStore() + hook_input = {"session_id": "s1", "cwd": "/tmp/repo", "last_assistant_message": "The parser uses UTC."} + for _ in range(2): + memory_core.record_user_prompt(store, {**hook_input, "prompt": "What timezone does the parser use?"}) + hook_runner.default_record_stop(store, hook_input) + # A tool notification and a flush must not make a repeated Stop look like a new answer. + store.record_event(repo(), "s1", "tool_result", {"tool": "Read"}) + packet, _ = store.prepare_flush(repo(), "s1", "session-end") + store.update_flush(packet, status="semantic-succeeded") + hook_runner.default_record_stop(store, hook_input) + + responses = store.conn.execute("SELECT payload_json FROM events WHERE kind = 'assistant_stop'").fetchall() + assert [json.loads(row[0])["text"] for row in responses] == ["The parser uses UTC."] * 2 + assert not store.has_unflushed_events(repo().identity, "s1") + store.close() + + +def test_concurrent_response_hooks_record_one_answer(isolated_env): + from concurrent.futures import ThreadPoolExecutor + from threading import Barrier + + store = memory_core.EvidenceStore() + store.record_event(repo(), "s1", "user_prompt", {"text": "Inspect the parser."}) + ready = Barrier(2) + + def capture(): + connection = memory_core.EvidenceStore(store.path) + try: + ready.wait(timeout=5) + connection.record_assistant_response(repo(), "s1", "The parser uses UTC.") + finally: + connection.close() + + with ThreadPoolExecutor(max_workers=2) as workers: + list(workers.map(lambda _: capture(), range(2))) + assert store.conn.execute("SELECT COUNT(*) FROM events WHERE kind = 'assistant_stop'").fetchone()[0] == 1 + store.record_assistant_response(repo(), "s1", "The serializer also uses UTC.") + store.record_assistant_response(repo(), "s2", "The parser uses UTC.") + assert store.conn.execute("SELECT COUNT(*) FROM events WHERE kind = 'assistant_stop'").fetchone()[0] == 3 + store.close() + + +def test_flush_worker_restores_harness_identity_from_child_environment(isolated_env, monkeypatch): + import flush_worker + + handoff_path = Path(os.environ["MEM0_CODE_DATA_DIR"]) / "pending" / "kimi.running" + handoff_path.parent.mkdir(parents=True, exist_ok=True) + handoff_path.write_text( + json.dumps({"hook_input": {"session_id": "s1", "cwd": "/tmp/repo"}}), + encoding="utf-8", + ) + monkeypatch.setenv("MEM0_PLUGIN_HARNESS", "kimi") + monkeypatch.setenv("MEM0_PLUGIN_ENV_PREFIX", "MEM0_KIMI") + monkeypatch.setenv("MEM0_PLUGIN_DATA_DIR_NAME", "kimi-plugin") + monkeypatch.setenv("MEM0_PLUGIN_SOURCE_TAG", "kimi_plugin") + monkeypatch.setattr("sys.argv", ["flush_worker.py", str(handoff_path)]) + + with ( + patch.object(flush_worker, "configure_harness", wraps=memory_core.configure_harness) as configure, + patch.object(flush_worker.telemetry, "init") as telemetry_init, + patch.object(flush_worker, "checkpoint_session", return_value={"status": "nothing-to-flush"}), + ): + flush_worker.main() + + configure.assert_called_once_with( + "kimi", + env_prefix="MEM0_KIMI", + data_dir_name="kimi-plugin", + source_tag="kimi_plugin", + ) + telemetry_init.assert_called_once_with(harness="kimi", source_tag="KIMI_PLUGIN") + memory_core.configure_harness( + "claude-code", data_dir_name="claude-code-plugin", source_tag="claude_code_plugin" + ) + + def test_launch_handoff_resolves_flush_worker_under_core(isolated_env): """hook.py lives in adapters/claude/; flush_worker.py lives in core/. @@ -1235,7 +1472,7 @@ def test_launch_handoff_resolves_flush_worker_under_core(isolated_env): sibling of hook.py, or every background flush launches against a nonexistent path. """ - import hook + import hook_runner data_dir_path = Path(os.environ["MEM0_CODE_DATA_DIR"]) pending_dir = data_dir_path / "pending" @@ -1243,13 +1480,18 @@ def test_launch_handoff_resolves_flush_worker_under_core(isolated_env): handoff_path = pending_dir / "test-handoff.json" handoff_path.write_text("{}", encoding="utf-8") - with patch.object(hook.subprocess, "Popen") as popen: - assert hook._launch_handoff(handoff_path) is True + with patch.object(hook_runner.subprocess, "Popen") as popen: + assert hook_runner._launch_handoff(handoff_path) is True launched_args = popen.call_args.args[0] worker_path = Path(launched_args[1]) assert worker_path == CORE / "flush_worker.py" assert worker_path.is_file() + child_env = popen.call_args.kwargs["env"] + assert child_env["MEM0_PLUGIN_HARNESS"] == "claude-code" + assert child_env["MEM0_PLUGIN_DATA_DIR_NAME"] == "claude-code-plugin" + assert child_env["MEM0_PLUGIN_SOURCE_TAG"] == "claude_code_plugin" + assert child_env["MEM0_CODE_DATA_DIR"] == os.environ["MEM0_CODE_DATA_DIR"] def test_flush_worker_detaches_on_windows_as_well_as_posix(monkeypatch): @@ -1258,30 +1500,30 @@ def test_flush_worker_detaches_on_windows_as_well_as_posix(monkeypatch): Without a creationflags fallback the worker stays attached to Claude's console on Windows and background extraction can die with the session. """ - import hook + import hook_runner - assert hook.detached_process_kwargs("darwin") == {"start_new_session": True} - assert hook.detached_process_kwargs("linux") == {"start_new_session": True} + assert hook_runner.detached_process_kwargs("darwin") == {"start_new_session": True} + assert hook_runner.detached_process_kwargs("linux") == {"start_new_session": True} monkeypatch.setattr(subprocess, "DETACHED_PROCESS", 0x00000008, raising=False) monkeypatch.setattr( subprocess, "CREATE_NEW_PROCESS_GROUP", 0x00000200, raising=False ) - windows_kwargs = hook.detached_process_kwargs("win32") + windows_kwargs = hook_runner.detached_process_kwargs("win32") assert windows_kwargs == {"creationflags": 0x00000208} assert "start_new_session" not in windows_kwargs def test_launch_handoff_always_requests_detachment(isolated_env): - import hook + import hook_runner pending_dir = Path(os.environ["MEM0_CODE_DATA_DIR"]) / "pending" pending_dir.mkdir(parents=True, exist_ok=True) handoff_path = pending_dir / "detach-handoff.json" handoff_path.write_text("{}", encoding="utf-8") - with patch.object(hook.subprocess, "Popen") as popen: - assert hook._launch_handoff(handoff_path) is True + with patch.object(hook_runner.subprocess, "Popen") as popen: + assert hook_runner._launch_handoff(handoff_path) is True assert popen.call_args.kwargs["start_new_session"] is True @@ -1294,23 +1536,23 @@ def test_pending_recovery_is_capped_and_leaves_the_rest_for_next_session( Each worker can block for MEM0_CODE_EXTRACTION_WAIT_SECONDS, so an uncapped launch is a thundering herd at session start. """ - import hook + import hook_runner pending_dir = Path(os.environ["MEM0_CODE_DATA_DIR"]) / "pending" pending_dir.mkdir(parents=True, exist_ok=True) - for number in range(hook.PENDING_LAUNCH_LIMIT + 3): + for number in range(hook_runner.PENDING_LAUNCH_LIMIT + 3): (pending_dir / f"packet-{number}.json").write_text("{}", encoding="utf-8") - with patch.object(hook.subprocess, "Popen"): - launched = hook.recover_pending_handoffs() + with patch.object(hook_runner.subprocess, "Popen"): + launched = hook_runner.recover_pending_handoffs() - assert launched == hook.PENDING_LAUNCH_LIMIT - assert len(list(pending_dir.glob("*.running"))) == hook.PENDING_LAUNCH_LIMIT + assert launched == hook_runner.PENDING_LAUNCH_LIMIT + assert len(list(pending_dir.glob("*.running"))) == hook_runner.PENDING_LAUNCH_LIMIT assert len(list(pending_dir.glob("*.json"))) == 3 def test_pending_recovery_drops_packets_past_the_expiry_window(isolated_env): - import hook + import hook_runner pending_dir = Path(os.environ["MEM0_CODE_DATA_DIR"]) / "pending" pending_dir.mkdir(parents=True, exist_ok=True) @@ -1319,11 +1561,11 @@ def test_pending_recovery_drops_packets_past_the_expiry_window(isolated_env): fresh = pending_dir / "fresh.json" fresh.write_text("{}", encoding="utf-8") - stale_time = time.time() - hook.PENDING_EXPIRY_SECONDS - 60 + stale_time = time.time() - hook_runner.PENDING_EXPIRY_SECONDS - 60 os.utime(expired, (stale_time, stale_time)) - with patch.object(hook.subprocess, "Popen"): - launched = hook.recover_pending_handoffs() + with patch.object(hook_runner.subprocess, "Popen"): + launched = hook_runner.recover_pending_handoffs() assert launched == 1 assert not expired.exists() @@ -2061,7 +2303,7 @@ def test_format_context_labels_memories_from_non_main_branches(): def test_sidekick_instructions_reject_unrequested_related_changes(): - prompt = (Path(__file__).parents[1] / "agents" / "sidekick.md").read_text() + prompt = (PLUGIN_ROOT / "agents" / "sidekick.md").read_text() normalized = " ".join(prompt.split()) assert "Skill" in prompt.split("---", 2)[1] assert "ALWAYS call `search_memories` before answering anything" in normalized @@ -2121,6 +2363,36 @@ def test_sidekick_reuses_parent_memory_once_and_records_lifecycle( store.close() +def test_sidekick_stop_without_agent_id_closes_latest_matching_run(isolated_env): + store = memory_core.EvidenceStore() + start_input = { + "session_id": "s1", + "cwd": "/tmp/repo", + "agent_id": "agent-123", + "agent_type": "sidekick", + } + + with patch.object(memory_core, "resolve_repo", return_value=repo()): + memory_core.record_sidekick_start(store, start_input) + memory_core.record_sidekick_stop( + store, + { + "session_id": "s1", + "cwd": "/tmp/repo", + "agent_type": "sidekick", + "last_assistant_message": "Done.", + }, + ) + + rows = store.conn.execute("SELECT agent_id, stopped_at, final_message FROM sidekick_runs").fetchall() + store.close() + + assert len(rows) == 1 + assert rows[0]["agent_id"] == "agent-123" + assert rows[0]["stopped_at"] + assert rows[0]["final_message"] == "Done." + + def test_search_once_per_session_avoids_repeated_remote_searches( isolated_env, monkeypatch ): @@ -2325,7 +2597,8 @@ def test_mcp_tool_contract_is_small_and_read_only(): "query", "top_k", "category", - "scope", "run_id", + "scope", + "run_id", } assert tool["inputSchema"]["required"] == ["query"] assert tool["inputSchema"]["properties"]["top_k"]["minimum"] == 1 @@ -2335,6 +2608,8 @@ def test_mcp_tool_contract_is_small_and_read_only(): ) assert tool["annotations"]["readOnlyHint"] is True assert tool["annotations"]["idempotentHint"] is True + assert "Claude" not in tool["description"] + assert "Claude" not in memory_core.PROJECT_MEMORY_INSTRUCTIONS def test_plugin_mcp_config_starts_the_local_server(): @@ -2430,6 +2705,51 @@ def test_mcp_tool_is_session_independent_and_returns_plain_memories( assert "project_knowledge" not in rendered +@pytest.mark.parametrize("scope", ["repo", "dir", "mine"]) +def test_mcp_search_passes_run_id_for_every_scope(isolated_env, monkeypatch, scope): + from dataclasses import replace + + monkeypatch.setenv("MEM0_API_KEY", "test-key") + project = replace(repo(), directory="src", project_id="code-example-hash") + monkeypatch.setattr(mcp_server, "resolve_repo", lambda cwd: project) + with patch.object( + memory_core, "_request_json_with_network_retry", return_value=({"results": []}, 0, 0) + ) as request: + mcp_server.call_search_memories({"query": "parser", "scope": scope}) + across_sessions = request.call_args.args[2]["filters"] + mcp_server.call_search_memories({"query": "parser", "scope": scope, "run_id": "session-42"}) + assert request.call_args.args[2]["filters"] == {"AND": [across_sessions, {"run_id": "session-42"}]} + + +def test_mcp_tool_uses_codex_workspace_metadata(isolated_env): + result = memory_core.MemorySearchResult( + succeeded=True, matched_count=0, already_shown_count=0, memories=[] + ) + request = { + "jsonrpc": "2.0", + "id": 2, + "method": "tools/call", + "params": { + "name": "search_memories", + "arguments": {"query": "parser"}, + "_meta": { + "x-codex-turn-metadata": { + "workspaces": {"/tmp/active-repository": {"has_changes": False}} + } + }, + }, + } + + with ( + patch.object(mcp_server, "resolve_repo", return_value=repo()) as resolve, + patch.object(mcp_server, "search_memories", return_value=result), + ): + response = mcp_server.handle_request(request) + + resolve.assert_called_once_with("/tmp/active-repository") + assert response["result"]["isError"] is False + + @pytest.mark.parametrize( "arguments, message", [ @@ -2438,7 +2758,8 @@ def test_mcp_tool_is_session_independent_and_returns_plain_memories( ({"query": "parser", "top_k": True}, "1 to 20"), ({"query": "parser", "category": "other"}, "supported categories"), ({"query": "parser", "threshold": 0.5}, "Unknown search argument"), - ({"query": "parser", "run_id": ""}, "run_id must be"), + ({"query": "parser", "run_id": " "}, "run_id must be a non-empty string"), + ({"query": "parser", "run_id": 42}, "run_id must be a non-empty string"), ], ) def test_mcp_tool_rejects_invalid_arguments(arguments, message): @@ -2447,7 +2768,7 @@ def test_mcp_tool_rejects_invalid_arguments(arguments, message): def test_search_skill_describes_memory_as_optional_starting_knowledge(): - prompt = (Path(__file__).parents[1] / "skills" / "search" / "SKILL.md").read_text() + prompt = (PLUGIN_ROOT / "skills" / "search" / "SKILL.md").read_text() normalized = " ".join(prompt.split()) assert "earlier work may already explain" in normalized @@ -2455,6 +2776,9 @@ def test_search_skill_describes_memory_as_optional_starting_knowledge(): assert "Call `search_memories` with the user's question" in normalized assert "Return the tool's result directly" in normalized assert "Run the search before repository exploration" not in normalized + assert "run_id" in normalized + assert "--run-id" in normalized + assert "Omit `run_id` to search across sessions" in normalized def test_control_skills_exposed(): @@ -2493,7 +2817,7 @@ def test_pause_skill_points_to_dedicated_resume_command(): def test_resume_skill_runs_cli_and_is_a_dedicated_command(): text = (PLUGIN_ROOT / "skills" / "resume" / "SKILL.md").read_text() assert "disable-model-invocation: true" in text - assert 'core/memory_cli.py" resume' in text + assert '--harness "claude-code" --plugin-data-dir "${CLAUDE_PLUGIN_DATA}" resume' in text def test_remember_skill_acknowledges_without_write_api(): @@ -2591,7 +2915,7 @@ def _run_hook( env.pop("CLAUDE_PLUGIN_OPTION_API_KEY", None) env.pop("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", None) return subprocess.run( - [sys.executable, str(ADAPTER / "hook.py"), action, *extra], + [sys.executable, str(ADAPTER), action, *extra], input=json.dumps(payload), text=True, capture_output=True, @@ -2774,7 +3098,7 @@ def test_first_user_prompt_searches_verbatim_and_returns_five_memories( for index in range(1, 7) ] - import hook + import hook_runner with ( patch.object(memory_core, "resolve_repo", return_value=repo()), @@ -2784,7 +3108,7 @@ def test_first_user_prompt_searches_verbatim_and_returns_five_memories( return_value=({"results": results}, 200, 600), ) as request, ): - output = hook.first_prompt_memory_output( + output = hook_runner.first_prompt_memory_output( store, {"session_id": "s1", "cwd": "/tmp/repo", "prompt": prompt}, ) @@ -2814,7 +3138,7 @@ def test_later_user_prompts_do_not_search_automatically(isolated_env, monkeypatc monkeypatch.setenv("MEM0_API_KEY", "m0-test-key") store = memory_core.EvidenceStore() - import hook + import hook_runner with ( patch.object(memory_core, "resolve_repo", return_value=repo()), @@ -2824,7 +3148,7 @@ def test_later_user_prompts_do_not_search_automatically(isolated_env, monkeypatc return_value=({"results": []}, 100, 20), ) as request, ): - first = hook.first_prompt_memory_output( + first = hook_runner.first_prompt_memory_output( store, { "session_id": "s1", @@ -2832,7 +3156,7 @@ def test_later_user_prompts_do_not_search_automatically(isolated_env, monkeypatc "prompt": "Where is ODS date formatting implemented?", }, ) - second = hook.first_prompt_memory_output( + second = hook_runner.first_prompt_memory_output( store, { "session_id": "s1", @@ -2853,7 +3177,7 @@ def test_manual_search_remains_available_after_automatic_search( monkeypatch.setenv("MEM0_API_KEY", "m0-test-key") store = memory_core.EvidenceStore() - import hook + import hook_runner with ( patch.object(memory_core, "resolve_repo", return_value=repo()), @@ -2878,7 +3202,7 @@ def test_manual_search_remains_available_after_automatic_search( ], ) as request, ): - hook.first_prompt_memory_output( + hook_runner.first_prompt_memory_output( store, { "session_id": "s1", @@ -2912,7 +3236,7 @@ def test_first_prompt_is_silent_when_all_matches_were_already_provided( } store.mark_injected("s1", repo().identity, [existing]) - import hook + import hook_runner with ( patch.object(memory_core, "resolve_repo", return_value=repo()), @@ -2922,7 +3246,7 @@ def test_first_prompt_is_silent_when_all_matches_were_already_provided( return_value=({"results": [existing]}, 100, 100), ), ): - output = hook.first_prompt_memory_output( + output = hook_runner.first_prompt_memory_output( store, { "session_id": "s1", @@ -2943,7 +3267,7 @@ def test_user_prompt_search_failure_still_records_evidence_and_emits_no_context( env["MEM0_API_KEY"] = "m0-test-key" env["MEM0_API_URL"] = "http://127.0.0.1:1" result = subprocess.run( - [sys.executable, str(ADAPTER / "hook.py"), "user-prompt"], + [sys.executable, str(ADAPTER), "user-prompt"], input=json.dumps( { "session_id": "s1", @@ -3027,7 +3351,7 @@ def test_plugin_entrypoints_share_explicit_claude_data_dir(tmp_path, monkeypatch sidekick = subprocess.run( [ sys.executable, - str(ADAPTER / "hook.py"), + str(ADAPTER), "sidekick-start", "--plugin-data-dir", str(canonical_data), @@ -3102,7 +3426,7 @@ def test_automatic_flush_can_be_disabled_for_external_harnesses(isolated_env): env["MEM0_API_KEY"] = "m0-test-key" start = subprocess.run( - [sys.executable, str(ADAPTER / "hook.py"), "user-prompt"], + [sys.executable, str(ADAPTER), "user-prompt"], input=json.dumps( {"session_id": "s1", "cwd": cwd, "prompt": "Inspect parser behavior."} ), @@ -3112,7 +3436,7 @@ def test_automatic_flush_can_be_disabled_for_external_harnesses(isolated_env): check=False, ) flushed = subprocess.run( - [sys.executable, str(ADAPTER / "hook.py"), "flush", "--reason", "session-end"], + [sys.executable, str(ADAPTER), "flush", "--reason", "session-end"], input=json.dumps({"session_id": "s1", "cwd": cwd}), text=True, capture_output=True, @@ -3136,12 +3460,29 @@ def test_automatic_flush_can_be_disabled_for_external_harnesses(isolated_env): def test_version_is_single_sourced(): manifest = json.loads((PLUGIN_ROOT / ".claude-plugin" / "plugin.json").read_text()) assert manifest["name"] == "mem0" - assert manifest["version"] == memory_core.PLUGIN_VERSION == "0.3.0" - root = PLUGIN_ROOT.parents[1] + assert manifest["version"] == memory_core.PLUGIN_VERSION == "0.3.1" + root = REPOSITORY_ROOT for mp in (root / "marketplace.json", root / ".claude-plugin" / "marketplace.json"): entry = next(p for p in json.loads(mp.read_text())["plugins"] if p["name"] == "mem0") assert entry["version"] == memory_core.PLUGIN_VERSION assert entry["source"] == "./integrations/claude-code-plugin" + manifests = { + "cursor": ".cursor-plugin/plugin.json", + "codex": ".codex-plugin/plugin.json", + "kimi": "kimi.plugin.json", + "mem0-agent": "plugin.json", + } + for host, filename in manifests.items(): + manifest = json.loads((root / "integrations" / f"{host}-plugin" / filename).read_text()) + assert manifest["version"] == memory_core.PLUGIN_VERSION + for host in ("claude-code", "cursor", "codex", "kimi", "antigravity"): + descriptor = json.loads((root / "integrations" / f"{host}-plugin" / "plugin-build.json").read_text()) + assert descriptor["version"] == memory_core.PLUGIN_VERSION + for directory in (".cursor-plugin", ".kimi-plugin"): + marketplace = json.loads((root / directory / "marketplace.json").read_text()) + assert marketplace["plugins"][0]["version"] == memory_core.PLUGIN_VERSION + response = mcp_server.handle_request({"jsonrpc": "2.0", "id": 1, "method": "initialize"}) + assert response["result"]["serverInfo"]["version"] == memory_core.PLUGIN_VERSION def test_user_id_falls_back_to_the_windows_account_name(monkeypatch): @@ -3183,7 +3524,7 @@ def test_transcript_rows_resume_from_a_byte_offset(tmp_path): ], ) - rows, end, resumed = memory_core._transcript_rows(str(transcript)) + rows, end, resumed = transcript_mod._transcript_rows(str(transcript)) assert [row["uuid"] for row in rows] == ["entry-1", "entry-2"] assert end == transcript.stat().st_size assert resumed is False @@ -3201,12 +3542,12 @@ def test_transcript_rows_resume_from_a_byte_offset(tmp_path): handle.write(json.dumps(third_row) + "\n") handle.write('{"uuid": "partial-row"') - rows, end, resumed = memory_core._transcript_rows(str(transcript), first_size) + rows, end, resumed = transcript_mod._transcript_rows(str(transcript), first_size) assert [row["uuid"] for row in rows] == ["entry-3"] assert resumed is True assert end < transcript.stat().st_size - rows, _, resumed = memory_core._transcript_rows( + rows, _, resumed = transcript_mod._transcript_rows( str(transcript), transcript.stat().st_size + 100 ) assert [row["uuid"] for row in rows] == ["entry-1", "entry-2", "entry-3"] @@ -3235,7 +3576,7 @@ def test_record_stop_reads_the_transcript_from_the_stored_offset( store = memory_core.EvidenceStore() with patch.object(memory_core, "resolve_repo", return_value=repo()): - memory_core.record_stop( + transcript_mod.record_stop( store, { "session_id": "s1", @@ -3270,7 +3611,7 @@ def test_record_stop_reads_the_transcript_from_the_stored_offset( }, ] _write_transcript(transcript, "s1", second_entries) - original_rows = memory_core._transcript_rows + original_rows = transcript_mod._transcript_rows offsets = [] def spying_rows(path, offset=0): @@ -3279,9 +3620,9 @@ def test_record_stop_reads_the_transcript_from_the_stored_offset( with ( patch.object(memory_core, "resolve_repo", return_value=repo()), - patch.object(memory_core, "_transcript_rows", side_effect=spying_rows), + patch.object(transcript_mod, "_transcript_rows", side_effect=spying_rows), ): - memory_core.record_stop( + transcript_mod.record_stop( store, { "session_id": "s1", @@ -3355,7 +3696,7 @@ def test_event_poll_touches_the_worker_heartbeat(isolated_env, monkeypatch, tmp_ def test_paused_session_start_keeps_pending_packets_fresh(isolated_env): - import hook + import hook_runner pending_dir = Path(os.environ["MEM0_CODE_DATA_DIR"]) / "pending" pending_dir.mkdir(parents=True, exist_ok=True) @@ -3363,11 +3704,11 @@ def test_paused_session_start_keeps_pending_packets_fresh(isolated_env): held.write_text("{}", encoding="utf-8") claimed = pending_dir / "claimed.running" claimed.write_text("{}", encoding="utf-8") - stale_time = time.time() - hook.PENDING_EXPIRY_SECONDS - 60 + stale_time = time.time() - hook_runner.PENDING_EXPIRY_SECONDS - 60 for path in (held, claimed): os.utime(path, (stale_time, stale_time)) - hook.refresh_pending_handoffs() + hook_runner.refresh_pending_handoffs() assert held.stat().st_mtime > time.time() - 60 assert claimed.stat().st_mtime > time.time() - 60 @@ -3635,7 +3976,7 @@ def test_forget_only_touches_shared_project_memory_when_asked(monkeypatch, isola repo = memory_core.RepoContext( cwd="/x", root="/x", identity="x", app_id="repo-a", branch="main", head_sha="abc", project_id="repo-a" ) - listed = [({"results": []}, 0, 0), ({"results": [{"id": "shared-1"}]}, 0, 0)] + listed = [({"results": []}, 0, 0), ({"results": [{"id": "shared-1", "app_id": "repo-a"}]}, 0, 0)] with patch.object(memory_core, "_request_json", side_effect=listed) as request: with patch.object(memory_core, "_delete_memory", return_value=True): result = memory_core.forget_remote_repo(repo, include_project_memory=True) @@ -3647,6 +3988,36 @@ def test_forget_only_touches_shared_project_memory_when_asked(monkeypatch, isola assert result == {"status": "deleted", "deleted": 1} +@pytest.mark.parametrize("include_project", [False, True]) +def test_forget_includes_legacy_shared_memories_only_when_requested(monkeypatch, isolated_env, include_project): + from dataclasses import replace + + monkeypatch.setenv("MEM0_API_KEY", "test-key") + project = replace(repo(), project_id="code-example-hash") + listed = { + "test-user": [{"id": "personal", "app_id": project.app_id}], + project.project_id: [{"id": "current", "app_id": project.app_id}], + project.app_id: [ + {"id": "legacy", "app_id": project.app_id}, + {"id": "other-app", "app_id": "another-repo"}, + ], + } + + def request(url, key, payload, timeout): + identity = next(iter(payload["filters"].values())) + return {"results": listed[identity]}, 0, 0 + + with ( + patch.object(memory_core, "_request_json", side_effect=request), + patch.object(memory_core, "_delete_memory", return_value=True) as delete, + ): + result = memory_core.forget_remote_repo(project, include_project_memory=include_project) + + expected = {"personal", "current", "legacy"} if include_project else {"personal"} + assert {call.args[2] for call in delete.call_args_list} == expected + assert result == {"status": "deleted", "deleted": len(expected)} + + def test_forget_reports_partial_failures(monkeypatch, isolated_env): monkeypatch.setenv("MEM0_API_KEY", "test-key") repo = memory_core.RepoContext( @@ -4025,7 +4396,8 @@ def test_same_named_folders_at_different_paths_get_different_namespaces(): b = memory_core._project_id("/home/raj/marketing", "local:/home/raj/marketing", "marketing") assert a != b assert a.startswith("local-marketing-") and b.startswith("local-marketing-") - assert memory_core._project_id("/x", "https://github.com/acme/api", "acme-api") == "acme-api" + remote_id = memory_core._project_id("/x", "https://github.com/acme/api", "acme-api") + assert remote_id.startswith("acme-api-") and remote_id != "acme-api" def test_relative_directory_is_empty_at_the_root_and_posix_below(): @@ -4171,3 +4543,88 @@ def test_every_write_carries_the_session_run_id(isolated_env, monkeypatch): assert len(bodies) == 1 assert bodies[0]["run_id"] == "s1" store.close() + + +def test_shared_transcripts_redact_without_truncating_before_extraction(): + secret = '"password": "plain-private-value" ' + prompt = secret + 'x' * 20000 + answer = secret + '界' * 40000 + events = [ + {"kind": "user_prompt", "payload": {"text": prompt}}, + {"kind": "assistant_stop", "payload": {"transcript_messages": [ + {"role": "user", "content": prompt}, + {"role": "assistant", "content": answer}, + ]}}, + ] + _, structured = memory_core.build_episode(repo(), "s1", "packet", events) + messages = memory_core.build_extraction_messages(structured) + assert [message["role"] for message in messages] == ["user", "assistant"] + assert "plain-private-value" not in json.dumps(structured) + assert [message["content"] for message in messages] == [memory_core.redact(prompt), memory_core.redact(answer)] + + +@pytest.mark.parametrize("status", [400, 401, 403, 404, 422]) +def test_event_polling_fails_fast_on_permanent_http_errors(status): + error = memory_core.urllib.error.HTTPError("https://example.test", status, "rejected", {}, None) + with patch.object(memory_core, "_get_json", side_effect=error), patch.object(memory_core.time, "sleep") as sleep: + with pytest.raises(memory_core.urllib.error.HTTPError): + memory_core._wait_for_event("https://example.test", "test-key", "event") + sleep.assert_not_called() + + +def test_prepare_flush_locks_before_reading_events(isolated_env): + store = memory_core.EvidenceStore() + store.record_event(repo(), "s1", "user_prompt", {"text": "Explain this."}) + statements = [] + store.conn.set_trace_callback(statements.append) + store.prepare_flush(repo(), "s1", "session-end") + assert statements[0] == "BEGIN IMMEDIATE" + assert not store.conn.in_transaction + store.close() + + +def test_extraction_batches_bound_individual_messages_without_losing_text(): + content = '界\\"\n' * 1000 + messages = [{"role": "assistant", "content": content}] + batches = memory_core.extraction_message_batches(messages, max_tokens=200) + assert all(memory_core._message_tokens(batch) <= 200 for batch in batches) + assert ''.join(message['content'] for batch in batches for message in batch) == content + + +def test_overlapping_subagent_stop_without_id_does_not_guess(isolated_env): + store = memory_core.EvidenceStore() + store.start_sidekick(repo(), "s1", "first", "sidekick", 0) + store.start_sidekick(repo(), "s1", "second", "sidekick", 0) + stopped = store.stop_sidekick(repo(), "s1", "", "sidekick", "", "Uncorrelated response") + assert stopped not in {"first", "second"} + rows = store.conn.execute("SELECT agent_id, stopped_at, final_message FROM sidekick_runs").fetchall() + assert all(row["stopped_at"] is None for row in rows if row["agent_id"] in {"first", "second"}) + assert next(row for row in rows if row["agent_id"] == stopped)["final_message"] == "Uncorrelated response" + store.close() + + +def test_delayed_handoff_survives_failed_atomic_rewrite(isolated_env, monkeypatch): + import flush_worker + + path = isolated_env / "checkpoint.running" + original = {"delay_seconds": 10, "hook_input": {"session_id": "s1"}} + path.write_text(json.dumps(original)) + monkeypatch.setattr(sys, "argv", ["flush_worker.py", str(path)]) + replace = Path.replace + + def fail_temporary(source, target): + if source.suffix == ".tmp": + raise OSError("simulated disk failure") + return replace(source, target) + + with patch.object(Path, "replace", fail_temporary), patch.object(flush_worker, "checkpoint_session") as checkpoint: + with pytest.raises(OSError, match="simulated disk failure"): + flush_worker.main() + assert json.loads(path.with_suffix(".json").read_text()) == original + assert not list(isolated_env.glob("*.tmp")) + checkpoint.assert_not_called() + + +def test_json_secret_redaction_handles_escaped_quotes(): + value = json.dumps({"password": 'prefix"private suffix'}) + assert json.loads(memory_core.redact(value)) == {"password": "[REDACTED]"} diff --git a/integrations/claude-code-plugin/tests/test_telemetry.py b/integrations/claude-code-plugin/tests/test_telemetry.py index a36980426..0ae280b19 100644 --- a/integrations/claude-code-plugin/tests/test_telemetry.py +++ b/integrations/claude-code-plugin/tests/test_telemetry.py @@ -7,9 +7,8 @@ from unittest.mock import patch import pytest - -PLUGIN_ROOT = Path(__file__).resolve().parents[1] -CORE = PLUGIN_ROOT / "core" +HOST_ROOT = Path(__file__).resolve().parents[1] +CORE = HOST_ROOT / "core" sys.path.insert(0, str(CORE)) import memory_core # noqa: E402 @@ -77,6 +76,46 @@ def test_record_hashes_identifiers_and_keeps_no_content(isolated_env): assert "session-abcdef" not in serialized +def test_record_rejects_sensitive_properties_at_the_shared_boundary(isolated_env): + secret = "sk-eval-12345678901234567890" + telemetry.record( + "search", + prompt=f"remember {secret}", + query=secret, + api_key=secret, + user_id="private-user", + note=f"failure contained {secret}", + memory_count=2, + ) + + (event,) = spool_lines() + assert event["properties"]["memory_count"] == 2 + serialized = json.dumps(event) + assert secret not in serialized + assert "private-user" not in serialized + assert not {"prompt", "query", "api_key", "user_id"} & event["properties"].keys() + + +@pytest.mark.parametrize("key", ["password", "token", "secret", "authorization"]) +def test_record_removes_sensitive_keys_from_nested_lists(isolated_env, key): + secret = "sk-eval-12345678901234567890" + telemetry.record( + "search", + details=[ + {key: "plain-value", "count": 2}, + {"nested": {key.upper(): "plain-value", "ok": True}}, + [f"failure contained {secret}"], + ], + ) + + (event,) = spool_lines() + assert event["properties"]["details"] == [ + {"count": 2}, + {"nested": {"ok": True}}, + ["failure contained [REDACTED]"], + ] + + def test_record_stops_appending_past_the_spool_cap(isolated_env): spool = memory_core.data_dir() / "telemetry.jsonl" spool.parent.mkdir(parents=True, exist_ok=True) diff --git a/integrations/codex-plugin/.codex-plugin/plugin.json b/integrations/codex-plugin/.codex-plugin/plugin.json new file mode 100644 index 000000000..916b72e66 --- /dev/null +++ b/integrations/codex-plugin/.codex-plugin/plugin.json @@ -0,0 +1,13 @@ +{ + "name": "mem0", + "version": "0.3.1", + "description": "Cross-session memory and token savings for coding agents.", + "author": { "name": "Mem0", "email": "support@mem0.ai" }, + "homepage": "https://docs.mem0.ai/integrations/codex", + "repository": "https://github.com/mem0ai/mem0", + "license": "Apache-2.0", + "keywords": ["memory", "coding-agents", "continual-learning", "token-efficiency"], + "skills": "./skills/", + "hooks": "./hooks/hooks.json", + "mcpServers": "./.mcp.json" +} diff --git a/integrations/codex-plugin/.mcp.json b/integrations/codex-plugin/.mcp.json new file mode 100644 index 000000000..00daa05bd --- /dev/null +++ b/integrations/codex-plugin/.mcp.json @@ -0,0 +1,19 @@ +{ + "mcpServers": { + "mem0": { + "type": "stdio", + "command": "python3", + "args": ["./core/mcp_server.py"], + "cwd": ".", + "env_vars": [ + "MEM0_API_KEY", + "MEM0_API_URL", + "MEM0_CODE_USER_ID", + "MEM0_USER_ID", + "MEM0_RESOLVED_USER_ID", + "MEM0_PROJECT_ID", + "MEM0_TELEMETRY" + ] + } + } +} diff --git a/integrations/codex-plugin/core/flush_worker.py b/integrations/codex-plugin/core/flush_worker.py new file mode 100644 index 000000000..6f7b9ecb2 --- /dev/null +++ b/integrations/codex-plugin/core/flush_worker.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Detached remote checkpoint worker. + +Claude Code may cancel SessionEnd hooks as a print-mode process exits. The hook +therefore persists its input first and launches this process in a new session. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + checkpoint_session, + configure_harness, + touch_handoff_heartbeat, +) + + +def main() -> int: + if len(sys.argv) != 2: + return 2 + handoff_path = Path(sys.argv[1]) + os.environ["MEM0_CODE_HANDOFF_PATH"] = str(handoff_path) + harness = os.environ.get("MEM0_PLUGIN_HARNESS") + if harness: + source_tag = os.environ.get("MEM0_PLUGIN_SOURCE_TAG", "") + configure_harness( + harness, + env_prefix=os.environ.get("MEM0_PLUGIN_ENV_PREFIX", ""), + data_dir_name=os.environ.get("MEM0_PLUGIN_DATA_DIR_NAME", ""), + source_tag=source_tag, + ) + telemetry.init(harness=harness, source_tag=source_tag.upper()) + completed = False + try: + payload = json.loads(handoff_path.read_text(encoding="utf-8")) + delay = float(payload.get("delay_seconds") or 0) + if delay > 0: + payload.pop("delay_seconds", None) + temporary = handoff_path.with_suffix(f".{os.getpid()}.tmp") + try: + temporary.write_text(json.dumps(payload), encoding="utf-8") + temporary.replace(handoff_path) + finally: + temporary.unlink(missing_ok=True) + time.sleep(delay) + if not handoff_path.exists(): + return 0 + hook_input = payload.get("hook_input") or {} + reason = str(payload.get("reason") or "checkpoint") + wait_for_inflight = bool(payload.get("wait_for_inflight")) + store = EvidenceStore() + try: + if wait_for_inflight: + session_id = str( + hook_input.get("session_id") or "unknown-session" + ) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + deadline = time.monotonic() + float( + os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120") + ) + while ( + store.has_inflight_flush(repo.identity, session_id) + and time.monotonic() < deadline + ): + touch_handoff_heartbeat() + time.sleep(0.25) + # Hooks capture the conversation before handoff; the worker only flushes it. + result = checkpoint_session(store, hook_input, reason) + print(json.dumps(result, sort_keys=True), flush=True) + completed = result.get("status") in { + "semantic-succeeded", + "explicitly-stored", + "nothing-to-flush", + } + finally: + store.close() + return 0 + finally: + telemetry.flush() + if completed: + try: + handoff_path.unlink() + except OSError: + pass + elif handoff_path.suffix == ".running": + try: + handoff_path.replace(handoff_path.with_suffix(".json")) + except OSError: + pass + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/codex-plugin/core/hook_runner.py b/integrations/codex-plugin/core/hook_runner.py new file mode 100644 index 000000000..2a2adb557 --- /dev/null +++ b/integrations/codex-plugin/core/hook_runner.py @@ -0,0 +1,372 @@ +"""Shared hook orchestration for all Mem0 agent plugins.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import subprocess +import sys +import time +import uuid +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + _session_id, + api_key, + bounded, + cache_plugin_api_key, + checkpoint_session, + clear_stale_api_key_cache, + configure_harness, + data_dir, + detached_process_kwargs, + format_context, + harness_config, + record_session_start, + record_tool, + record_user_prompt, + redact, + search_memories, +) + +STALE_RUNNING_SECONDS = 300 +PENDING_EXPIRY_SECONDS = 7 * 24 * 60 * 60 +PENDING_LAUNCH_LIMIT = 5 +DEFAULT_IDLE_FLUSH_SECONDS = 300 + +_core_dir: Path = Path(__file__).resolve().parent + + +def read_hook_input() -> dict: + try: + value = json.load(sys.stdin) + return value if isinstance(value, dict) else {} + except (json.JSONDecodeError, OSError): + return {} + + +def default_record_stop(store: EvidenceStore, hook_input: dict): + """Record the assistant's response without transcript parsing.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + message = redact(hook_input.get("last_assistant_message", "")).strip() + if message: + store.record_assistant_response(repo, session_id, message) + return repo, session_id + + +def first_prompt_memory_output(store: EvidenceStore, hook_input: dict) -> dict: + """Search once before the agent handles the first prompt in a session.""" + repo, session_id, prompt, is_first_prompt = record_user_prompt(store, hook_input) + if not is_first_prompt: + return {} + try: + minimum_query_chars = int(os.environ.get("MEM0_CODE_MIN_QUERY_CHARS", "20")) + except ValueError: + minimum_query_chars = 20 + if len(prompt.strip()) < max(minimum_query_chars, 1): + return {} + result = search_memories( + store, repo, session_id, bounded(prompt, 6000), + top_k=5, operation="first-prompt-search", timeout=2, + ) + if not result.memories: + return {} + context = format_context( + result.memories, + "Mem0 found these relevant memories from earlier work in this repository:", + ) + telemetry.record( + "context_injected", + repo=repo, session_id=session_id, trigger="first-prompt", + memory_count=len(result.memories), context_chars=len(context), + prompt_chars=len(prompt), + ) + return { + "hookSpecificOutput": { + "hookEventName": "UserPromptSubmit", + "additionalContext": context, + }, + } + + +def _launch_handoff(handoff_path: Path) -> bool: + running_path = handoff_path.with_suffix(".running") + try: + handoff_path.replace(running_path) + except OSError: + return False + worker = _core_dir / "flush_worker.py" + log_path = data_dir() / "flush-worker.log" + log_handle = open(log_path, "a", encoding="utf-8") + harness = harness_config() + child_env = os.environ.copy() + child_env.update( + { + "MEM0_CODE_DATA_DIR": str(data_dir()), + "MEM0_PLUGIN_HARNESS": harness["name"], + "MEM0_PLUGIN_ENV_PREFIX": harness["env_prefix"], + "MEM0_PLUGIN_DATA_DIR_NAME": harness["data_dir_name"], + "MEM0_PLUGIN_SOURCE_TAG": harness["source_tag"], + } + ) + try: + subprocess.Popen( + [sys.executable, str(worker), str(running_path)], + stdin=subprocess.DEVNULL, + stdout=log_handle, stderr=log_handle, + close_fds=True, + env=child_env, + **detached_process_kwargs(), + ) + finally: + log_handle.close() + return True + + +def recover_pending_handoffs() -> int: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + now = time.time() + for running in pending_dir.glob("*.running"): + try: + if now - running.stat().st_mtime > STALE_RUNNING_SECONDS: + running.replace(running.with_suffix(".json")) + except OSError: + continue + recoverable = [] + for handoff in pending_dir.glob("*.json"): + try: + age = now - handoff.stat().st_mtime + except OSError: + continue + if age > PENDING_EXPIRY_SECONDS: + handoff.unlink(missing_ok=True) + continue + recoverable.append((age, handoff)) + recoverable.sort(key=lambda item: item[0], reverse=True) + launched = 0 + for _, handoff in recoverable[:PENDING_LAUNCH_LIMIT]: + launched += int(_launch_handoff(handoff)) + return launched + + +def refresh_pending_handoffs() -> None: + pending_dir = data_dir() / "pending" + if not pending_dir.is_dir(): + return + for pattern in ("*.json", "*.running"): + for handoff in pending_dir.glob(pattern): + try: + os.utime(handoff) + except OSError: + continue + + +def hand_off_flush( + hook_input: dict, reason: str, *, wait_for_inflight: bool = False, +) -> None: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = ( + f"{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}\0{reason}" + ) + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + handoff_path = pending_dir / f"{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": reason, + "wait_for_inflight": wait_for_inflight, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + + +def automatic_flush_enabled() -> bool: + return os.environ.get("MEM0_CODE_AUTO_FLUSH", "true").lower() in { + "1", "true", "yes", "on", + } + + +def schedule_periodic_checkpoint( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + if ( + not automatic_flush_enabled() + or not api_key() + or not store.checkpoint_due(repo.identity, session_id) + ): + return False + if store.prepare_flush(repo, session_id, "periodic") is None: + return False + hand_off_flush(hook_input, "periodic") + return True + + +def _idle_flush_seconds() -> int: + try: + return max( + int(os.environ.get("MEM0_CODE_IDLE_FLUSH_SECONDS", str(DEFAULT_IDLE_FLUSH_SECONDS))), + 0, + ) + except ValueError: + return DEFAULT_IDLE_FLUSH_SECONDS + + +def schedule_idle_flush( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + delay = _idle_flush_seconds() + if delay <= 0 or not automatic_flush_enabled() or not api_key(): + return False + if store.has_inflight_flush(repo.identity, session_id): + return False + if not store.has_unflushed_events(repo.identity, session_id): + return False + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = f"idle\0{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}" + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + for old in pending_dir.glob(f"idle-{digest}*"): + old.unlink(missing_ok=True) + handoff_path = pending_dir / f"idle-{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": "idle", + "delay_seconds": delay, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + return True + + +def log_failure(exc: Exception) -> None: + try: + log_path = data_dir() / "plugin-errors.log" + with log_path.open("a", encoding="utf-8") as handle: + handle.write(f"{time.time():.3f} {type(exc).__name__}: {exc}\n") + except OSError: + pass + + +def run( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> int: + if record_stop_fn is None: + record_stop_fn = default_record_stop + if automatic_flush_reasons is None: + automatic_flush_reasons = {"session-end"} + + base_actions = ["session-start", "user-prompt", "post-tool", "stop", "flush"] + all_actions = base_actions + list((extra_actions or {}).keys()) + + parser = argparse.ArgumentParser() + parser.add_argument("action", choices=all_actions) + parser.add_argument("--reason", default="manual") + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + args = parser.parse_args() + + if args.harness: + configure_harness(args.harness) + telemetry.init(harness=args.harness) + + if args.plugin_data_dir: + os.environ[data_dir_env] = args.plugin_data_dir + + cache_plugin_api_key() + if args.action == "session-start": + clear_stale_api_key_cache() + + hook_input = read_hook_input() + store = EvidenceStore() + try: + if store.is_paused(): + if args.action == "session-start": + refresh_pending_handoffs() + telemetry.record("session_start", paused=True) + telemetry.spawn_flush() + return 0 + + if args.action == "session-start": + if telemetry.is_first_run(): + telemetry.record("install") + recovered = recover_pending_handoffs() + record_session_start(store, hook_input) + if recovered: + telemetry.record("handoff_recovered", count=recovered) + telemetry.spawn_flush() + elif args.action == "user-prompt": + output = first_prompt_memory_output(store, hook_input) + if output: + print(json.dumps(output)) + elif args.action == "post-tool": + record_tool(store, hook_input) + elif args.action == "stop": + repo, session_id = record_stop_fn(store, hook_input) + if not schedule_periodic_checkpoint(store, hook_input, repo, session_id): + schedule_idle_flush(store, hook_input, repo, session_id) + elif args.action == "flush": + automatic = args.reason in automatic_flush_reasons + if automatic and not automatic_flush_enabled(): + return 0 + if args.reason == "session-end": + record_stop_fn(store, hook_input) + if os.environ.get("MEM0_CODE_SYNC_FLUSH") == "1": + print(json.dumps(checkpoint_session(store, hook_input, args.reason))) + else: + session_id = str(hook_input.get("session_id") or "unknown-session") + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + already_running = store.has_inflight_flush(repo.identity, session_id) + if already_running and args.reason == "session-end": + hand_off_flush(hook_input, args.reason, wait_for_inflight=True) + elif not already_running and store.prepare_flush( + repo, session_id, args.reason, + ) is not None: + hand_off_flush(hook_input, args.reason) + elif extra_actions and args.action in extra_actions: + result = extra_actions[args.action](store, hook_input) + if result: + print(json.dumps(result)) + finally: + store.close() + return 0 + + +def entry_point( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> None: + try: + raise SystemExit(run( + record_stop_fn=record_stop_fn, + extra_actions=extra_actions, + data_dir_env=data_dir_env, + automatic_flush_reasons=automatic_flush_reasons, + )) + except Exception as exc: + log_failure(exc) + raise SystemExit(0) + + +if __name__ == "__main__": + entry_point() diff --git a/integrations/codex-plugin/core/mcp_server.py b/integrations/codex-plugin/core/mcp_server.py new file mode 100644 index 000000000..036fbbdc9 --- /dev/null +++ b/integrations/codex-plugin/core/mcp_server.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Expose Mem0's memory search as one local coding-agent tool.""" + +from __future__ import annotations + +import json +import os +import sys +from typing import Any + +import telemetry +from memory_core import ( + CODING_MEMORY_CATEGORY_NAMES, + PLUGIN_VERSION, + SEARCH_SCOPES, + format_search_result, + resolve_repo, + search_memories, +) + +PROTOCOL_VERSION = "2024-11-05" +TOOL_NAME = "search_memories" +TOOL_DESCRIPTION = ( + "Search memories from earlier work in this repository. ALWAYS call this " + "tool before answering anything that could depend on prior context: the " + "user's preferences, facts about this codebase, history, people, projects, " + "or earlier decisions. Do not rely on the chat window alone. The " + "repository's memory is shared by everyone who works in it and includes " + "what it took to run, test, or build here, so search before assuming an " + "invocation works. The scope argument changes what is searched: 'repo' " + "(default) is the whole repository's shared memory plus your own " + "preferences, 'dir' narrows the shared part to the directory you are " + "working in, and 'mine' is your preferences alone." +) +TOOL_SCHEMA = { + "type": "object", + "properties": { + "query": { + "type": "string", + "minLength": 1, + "maxLength": 2000, + "description": "A direct question about earlier work in this repository.", + }, + "top_k": { + "type": "integer", + "minimum": 1, + "maximum": 20, + "description": "Maximum memories to return. Uses Mem0's configured default when omitted.", + }, + "category": { + "type": "string", + "enum": list(CODING_MEMORY_CATEGORY_NAMES), + "description": "Optional memory category. Omit to search every category.", + }, + "scope": { + "type": "string", + "enum": list(SEARCH_SCOPES), + "description": ( + "Which memories to search. 'repo' (default) is the whole repository's " + "shared memory plus your own preferences, 'dir' narrows the shared " + "part to the current directory, 'mine' is your preferences alone." + ), + }, + "run_id": { + "type": "string", + "minLength": 1, + "description": ( + "Optional coding-agent session ID. With any scope, restricts results to memories " + "saved in that session. Omit to recall memories across sessions." + ), + }, + }, + "required": ["query"], + "additionalProperties": False, +} + + +class ToolInputError(ValueError): + pass + + +def _validate_arguments( + arguments: Any, +) -> tuple[str, int | None, str | None, str | None, str | None]: + if not isinstance(arguments, dict): + raise ToolInputError("Search arguments must be an object.") + + unknown = set(arguments) - {"query", "top_k", "category", "scope", "run_id"} + if unknown: + raise ToolInputError(f"Unknown search argument: {sorted(unknown)[0]}") + + query = arguments.get("query") + if not isinstance(query, str) or not query.strip(): + raise ToolInputError("query must be a non-empty string.") + query = query.strip() + if len(query) > 2000: + raise ToolInputError("query must be at most 2,000 characters.") + + top_k = arguments.get("top_k") + if top_k is not None and ( + isinstance(top_k, bool) or not isinstance(top_k, int) or not 1 <= top_k <= 20 + ): + raise ToolInputError("top_k must be an integer from 1 to 20.") + + category = arguments.get("category") + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ToolInputError("category must be one of Mem0's supported categories.") + + scope = arguments.get("scope") + if scope is not None and scope not in SEARCH_SCOPES: + raise ToolInputError(f"scope must be one of {list(SEARCH_SCOPES)}.") + + run_id = arguments.get("run_id") + if run_id is not None: + if not isinstance(run_id, str) or not run_id.strip(): + raise ToolInputError("run_id must be a non-empty string.") + run_id = run_id.strip() + + return query, top_k, category, scope, run_id + + +def call_search_memories(arguments: Any, cwd: str | None = None) -> str: + query, top_k, category, scope, run_id = _validate_arguments(arguments) + repo = resolve_repo(cwd or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + result = search_memories( + None, + repo, + None, + query, + top_k=top_k, + category=category, + scope=scope, + run_id=run_id, + operation="mcp-search", + ) + return format_search_result(result) + + +def _workspace_cwd(params: dict[str, Any]) -> str | None: + meta = params.get("_meta") + if not isinstance(meta, dict): + return None + metadata = meta.get("x-codex-turn-metadata") + if not isinstance(metadata, dict): + return None + workspaces = metadata.get("workspaces") or {} + if isinstance(workspaces, dict): + return next((path for path in workspaces if isinstance(path, str) and path), None) + return None + + +def _tool_response(text: str, *, is_error: bool = False) -> dict[str, Any]: + return { + "content": [{"type": "text", "text": text}], + "isError": is_error, + } + + +def handle_request(message: Any) -> dict[str, Any] | None: + if not isinstance(message, dict): + return None + request_id = message.get("id") + method = message.get("method") + + if method == "notifications/initialized": + return None + if method == "initialize": + requested = (message.get("params") or {}).get("protocolVersion") + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "protocolVersion": requested or PROTOCOL_VERSION, + "capabilities": {"tools": {"listChanged": False}}, + "serverInfo": {"name": "mem0", "version": PLUGIN_VERSION}, + }, + } + if method == "ping": + return {"jsonrpc": "2.0", "id": request_id, "result": {}} + if method == "tools/list": + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "tools": [ + { + "name": TOOL_NAME, + "description": TOOL_DESCRIPTION, + "inputSchema": TOOL_SCHEMA, + "annotations": { + "readOnlyHint": True, + "idempotentHint": True, + "openWorldHint": True, + }, + } + ] + }, + } + if method == "tools/call": + params = message.get("params") or {} + if params.get("name") != TOOL_NAME: + result = _tool_response("Unknown Mem0 tool.", is_error=True) + else: + try: + result = _tool_response( + call_search_memories(params.get("arguments"), _workspace_cwd(params)) + ) + except ToolInputError as exc: + result = _tool_response(str(exc), is_error=True) + except Exception: + result = _tool_response("Memory search failed.", is_error=True) + return {"jsonrpc": "2.0", "id": request_id, "result": result} + if request_id is None: + return None + return { + "jsonrpc": "2.0", + "id": request_id, + "error": {"code": -32601, "message": "Method not found"}, + } + + +def main() -> int: + for raw_line in sys.stdin: + try: + message = json.loads(raw_line) + response = handle_request(message) + except json.JSONDecodeError: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32700, "message": "Parse error"}, + } + except Exception: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32603, "message": "Internal error"}, + } + if response is not None: + sys.stdout.write(json.dumps(response, separators=(",", ":")) + "\n") + sys.stdout.flush() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/codex-plugin/core/memory_cli.py b/integrations/codex-plugin/core/memory_cli.py new file mode 100644 index 000000000..595729feb --- /dev/null +++ b/integrations/codex-plugin/core/memory_cli.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Mem0 diagnostics and user controls.""" + +from __future__ import annotations + +import argparse +import json +import os + +import telemetry +from memory_core import ( + EvidenceStore, + api_key, + data_dir, + doctor, + forget_remote_repo, + configure_harness, + resolve_repo, + user_id, +) + + +def _print_status(value: dict) -> None: + last = value.get("last_operation") or {} + print(f"Mem0: {'paused' if value['paused'] else 'active'}") + print(f"Repository: {value['repo_id']}") + print(f"Local data: {value['data_dir']}") + print(f"API key: {'configured' if value['api_key_configured'] else 'missing'}") + print( + "Saved on this computer: " + f"{value['events']} session details, {value['flushes']} memory updates" + ) + print( + f"Used in this repository: {value['retrievals']} memories returned, " + f"{value['sidekick_runs']} sidekick runs" + ) + if last: + item_label = "" + if last["operation"] in {"flush", "flush-retry"}: + item_label = f", {last['item_count']} memories" + operation = ( + "memory update" + if last["operation"] in {"flush", "flush-retry"} + else last["operation"].replace("-", " ") + ) + print( + f"Last {operation}: " + f"{'succeeded' if last['success'] else 'failed'} " + f"({last['duration_ms']:.1f} ms{item_label})" + ) + sidekick = value.get("last_sidekick") or {} + if sidekick: + state = "finished" if sidekick.get("stopped_at") else "started" + print( + "Last sidekick: " + f"{state}, received {sidekick['context_chars']} characters of memory, " + f"agent {sidekick['agent_id']}" + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + subparsers = parser.add_subparsers(dest="command", required=True) + + status = subparsers.add_parser("status") + status.add_argument("--json", action="store_true") + + doctor_parser = subparsers.add_parser("doctor") + doctor_parser.add_argument("--json", action="store_true") + + subparsers.add_parser("pause") + subparsers.add_parser("resume") + + forget = subparsers.add_parser("forget") + forget.add_argument("--remote", action="store_true") + forget.add_argument("--yes", action="store_true") + forget.add_argument("--include-project-memory", action="store_true") + + args = parser.parse_args() + if args.harness: + source_tag = f"{args.harness.replace('-', '_')}_plugin" + configure_harness(args.harness, source_tag=source_tag) + telemetry.init(harness=args.harness, source_tag=source_tag.upper()) + if args.plugin_data_dir: + os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir + store = EvidenceStore() + try: + repo = resolve_repo(os.getcwd()) + telemetry.record("control", repo=repo, action=args.command) + if args.command == "status": + result = { + **store.status(repo.identity), + "repo_id": repo.identity, + "app_id": repo.app_id, + "project_id": repo.project_id, + "directory": repo.directory, + "user_id": user_id(), + "data_dir": str(data_dir()), + "api_key_configured": bool(api_key()), + } + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + _print_status(result) + elif args.command == "doctor": + result = doctor(os.getcwd()) + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + for name, check in result["checks"].items(): + print( + f"{'PASS' if check['ok'] else 'FAIL'} {name}: {check['detail']}" + ) + return 0 if result["ok"] else 1 + elif args.command == "pause": + store.set_setting("paused", "true") + print("Mem0 stopped saving and searching memories.") + elif args.command == "resume": + store.set_setting("paused", "false") + print("Mem0 resumed saving and searching memories.") + elif args.command == "forget": + if not args.yes: + print( + "Refusing to delete data without --yes. Add --remote to also " + "delete this user/repository scope from Mem0." + ) + return 2 + remote_result = ( + forget_remote_repo( + repo, include_project_memory=args.include_project_memory + ) + if args.remote + else None + ) + local_result = store.forget_local_repo(repo.identity) + print( + json.dumps( + {"local": local_result, "remote": remote_result}, + indent=2, + default=str, + ) + ) + if remote_result and remote_result.get("status") == "error": + return 1 + finally: + store.close() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/codex-plugin/core/memory_core.py b/integrations/codex-plugin/core/memory_core.py new file mode 100644 index 000000000..cf71196b8 --- /dev/null +++ b/integrations/codex-plugin/core/memory_core.py @@ -0,0 +1,2645 @@ +#!/usr/bin/env python3 +"""Shared core for Mem0 agent plugins. + +Hooks record small session details locally. When the agent compacts or ends the +session, Mem0 sends the useful parts to the platform so it can create memories. +The agent can search those memories during later work in the repository. +""" + +from __future__ import annotations + +import functools +import hashlib +import json +import math +import os +import re +import sqlite3 +import subprocess +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +import telemetry + +DEFAULT_API_URL = "https://api.mem0.ai" +PLUGIN_VERSION = "0.3.1" + +_harness_name: str = "generic" +_harness_env_prefix: str = "MEM0_PLUGIN" +_harness_data_dir_name: str = "mem0-plugin" +_harness_source_tag: str = "mem0_plugin" + + +def configure_harness( + name: str, + env_prefix: str = "", + data_dir_name: str = "", + source_tag: str = "", +) -> None: + global _harness_name, _harness_env_prefix, _harness_data_dir_name, _harness_source_tag + _harness_name = name + _harness_env_prefix = env_prefix or f"MEM0_{name.upper().replace('-', '_')}" + _harness_data_dir_name = data_dir_name or f"{name}-plugin" + _harness_source_tag = source_tag or f"{name.replace('-', '_')}_plugin" + + +def harness_config() -> dict[str, str]: + return { + "name": _harness_name, + "env_prefix": _harness_env_prefix, + "data_dir_name": _harness_data_dir_name, + "source_tag": _harness_source_tag, + } + + +MAX_COMMAND_CHARS = 2000 +MAX_RESULT_CHARS = 2500 +MAX_EPISODE_CHARS = 12000 +CHECKPOINT_EXCHANGES = 5 +CHECKPOINT_MESSAGES = 10 +CHECKPOINT_SOURCE_CHARS = 40000 +DEFAULT_MAX_CONTEXT_CHARS = 4000 +MAX_EXTRACTION_INPUT_TOKENS = 24000 +MAX_FLUSH_ATTEMPTS = 5 +FORGET_PAGE_SIZE = 100 +FORGET_MAX_PAGES = 50 + +PROJECT_MEMORY_INSTRUCTIONS = """Save concise repository facts that will help anyone with future coding work in this repository. + +A completed change should produce one memory explaining the resulting behavior, where it is implemented when useful, and any important constraints or reasoning. Exploration or accepted decisions may produce separate memories only when they are independently useful. + +A command that failed and was then made to work should produce one memory naming the failing invocation, the error it returned, and the invocation that succeeded. Do not save one-off errors caused by an edit still in progress, transient network failures, or anything a rerun would fix on its own. + +Use the current coding agent's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or the coding agent completed them. Treat subagent responses as supporting repository evidence, not as decisions. + +Write about the repository, not the user, assistant, session, or task. Do not save personal preferences. Do not save a memory that only states which repository, branch, or directory the session worked in. Do not include test results, documentation updates, release notes, or temporary state. + +If nothing useful was established, return no memories.""" + +PERSONAL_MEMORY_INSTRUCTIONS = """Save concise facts about the user that will help in any repository: preferred tools, package managers, languages, coding style, review and communication preferences, and anything the user explicitly asked to be remembered about themselves. + +Write in the third person about the user, not about the repository, the assistant, the session, or the task. Do not save repository facts, project decisions, commands, or what was built. + +Never save that the user has no preferences or that nothing was learned. If nothing was learned about the user, return no memories.""" + +CODING_MEMORY_CATEGORIES = [ + { + "project_knowledge": ( + "What the project is and how its code, APIs, data, files, and " + "components work." + ) + }, + { + "decisions_and_constraints": ( + "Why an approach was chosen, what must remain true, and rules future " + "work must follow." + ) + }, + { + "workflows": ( + "How to run, test, debug, deploy, configure, or otherwise work on the " + "project." + ) + }, + { + "problems_and_fixes": ( + "Bugs, failures, known pitfalls, their causes, and how to fix or avoid " + "them." + ) + }, + { + "results": ( + "Outcomes and measurements from tests, benchmarks, experiments, or " + "investigations." + ) + }, +] +CODING_MEMORY_CATEGORY_NAMES = tuple( + category_name + for category in CODING_MEMORY_CATEGORIES + for category_name in category +) + +TEST_COMMAND_RE = re.compile( + r"(?:^|\s)(?:pytest|py\.test|jest|vitest|go\s+test|cargo\s+test|" + r"npm\s+(?:run\s+)?test|pnpm\s+(?:run\s+)?test|yarn\s+test|" + r"mvn\s+test|gradle\s+test|make\s+test)(?:\s|$)", + re.IGNORECASE, +) +BUILD_COMMAND_RE = re.compile( + r"(?:^|\s)(?:npm|pnpm|yarn)\s+(?:run\s+)?build(?:\s|$)|" + r"(?:^|\s)(?:cargo|go|mvn|gradle|make)\s+build(?:\s|$)", + re.IGNORECASE, +) + +SECRET_PATTERNS = [ + re.compile(r"(?i)(authorization\s*[:=]\s*(?:bearer|token)\s+)[^\s\"']+"), + re.compile( + r"(?i)((?:api[_-]?key|secret[_-]?access[_-]?key|session[_-]?token)\s*[:=]\s*)[^\s\"']+" + ), + re.compile( + r"(?i)((?:access[_-]?token|refresh[_-]?token|password|credential)" + r"\s*[:=]\s*)[^\s&\"']+" + ), + re.compile(r"\b(?:sk|m0|mem0_sk|psk)-[A-Za-z0-9_\-]{12,}\b"), + re.compile(r"\b(?:ASIA|AKIA)[A-Z0-9]{12,}\b"), + re.compile(r"\b(?:ghp_|github_pat_|xox[baprs]-)[A-Za-z0-9_\-]{12,}\b"), + re.compile( + r"-----BEGIN [^-]*PRIVATE KEY-----.*?-----END [^-]*PRIVATE KEY-----", + re.DOTALL, + ), + re.compile( + r'(?i)("(?:api[_-]?key|password|secret(?:[_-]?access[_-]?key)?' + r'|(?:access|refresh|session)[_-]?token|token|authorization|credential' + r')"\s*:\s*")(?:\\.|[^"\\])*' + ), +] + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def redact(value: Any) -> str: + text = ( + value + if isinstance(value, str) + else json.dumps(value, ensure_ascii=False, default=str) + ) + for pattern in SECRET_PATTERNS: + if pattern.groups: + text = pattern.sub(r"\1[REDACTED]", text) + else: + text = pattern.sub("[REDACTED]", text) + return text + + +def bounded(value: Any, limit: int) -> str: + text = redact(value).strip() + if len(text) <= limit: + return text + return text[:limit] + f"\n...[truncated {len(text) - limit} chars]" + + +def _git(cwd: str, *args: str) -> str: + try: + result = subprocess.run( + ["git", "-C", cwd, *args], + check=False, + capture_output=True, + text=True, + timeout=0.5, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return result.stdout.strip() if result.returncode == 0 else "" + + +def _normalize_remote(remote: str) -> str: + remote = remote.strip() + if remote.startswith("git@") and ":" in remote: + host_path = remote[4:].replace(":", "/", 1) + remote = f"https://{host_path}" + if remote.endswith(".git"): + remote = remote[:-4] + if "://" in remote: + parsed = urllib.parse.urlsplit(remote) + hostname = parsed.hostname or "" + if parsed.port: + hostname = f"{hostname}:{parsed.port}" + remote = urllib.parse.urlunsplit( + (parsed.scheme, hostname, parsed.path, parsed.query, parsed.fragment) + ) + return remote.rstrip("/") + + +_WILDCARD_SCOPE = re.compile(r"^\*+$") + + +def _scope_value(raw: str | None) -> str: + """Reject wildcards as identities: they are filter syntax and would widen the scope.""" + value = (raw or "").strip() + return "" if _WILDCARD_SCOPE.match(value) else value + + +SEARCH_SCOPES = ("repo", "dir", "mine") +DEFAULT_SEARCH_SCOPE = "repo" + +def directory_app_id(repo: RepoContext) -> str: + """The app_id of the directory this session runs in: the repository at the root, repository/path below it.""" + return f"{repo.app_id}/{repo.directory}" if repo.directory else repo.app_id + + +def directory_chain(repo: RepoContext) -> list[str]: + """Every directory a memory belongs to, from the top-level folder down to the one it was written in.""" + parts = repo.directory.split("/") if repo.directory else [] + return ["/".join(parts[: index + 1]) for index in range(len(parts))] + + +def _shared_project_ids(repo: RepoContext) -> list[str]: + """Current and pre-upgrade namespaces, shared by recall and explicit deletion.""" + if not repo.identity.startswith("local:") and repo.project_id != repo.app_id: + return [repo.project_id, repo.app_id] + return [repo.project_id] + + +def _search_filters(user: str, repo: RepoContext, scope: str) -> dict[str, Any]: + """Build the scope filter: app_id scopes to the repo, then union shared and personal lanes.""" + app_scope = {"app_id": repo.app_id} + mine = {"AND": [{"user_id": user}, app_scope]} + if scope == "mine": + return mine + projects = [{"AND": [{"agent_id": project_id}, app_scope]} for project_id in _shared_project_ids(repo)] + shared: dict[str, Any] = projects[0] if len(projects) == 1 else {"OR": projects} + if scope == "dir" and repo.directory: + shared = {"AND": [shared, {"metadata": {"dirs": {"contains": repo.directory}}}]} + return {"OR": [shared, mine]} + + +def search_scope() -> str: + configured = ( + _plugin_option("search_scope", "MEM0_CODE_SEARCH_SCOPE") or "" + ).strip().lower() + return configured if configured in SEARCH_SCOPES else DEFAULT_SEARCH_SCOPE + + +def resolve_search_scope(scope: str | None) -> str: + value = (scope or search_scope()).strip().lower() + if value not in SEARCH_SCOPES: + raise ValueError(f"Unknown search scope: {value}") + return value + + +def _legacy_project_map(cwd: str, root: str, raw_remote: str) -> str: + """Return the project name used by the previous Claude Code plugin.""" + try: + data = json.loads((Path.home() / ".mem0" / "project_map.json").read_text()) + except (OSError, json.JSONDecodeError): + return "" + if not isinstance(data, dict): + return "" + + keys = list(dict.fromkeys([cwd, root, os.path.realpath(cwd), os.path.realpath(root)])) + if raw_remote: + keys.append(f"remote:{hashlib.sha256(raw_remote.encode()).hexdigest()[:16]}") + for key in keys: + value = data.get(key) + if isinstance(value, str) and value.strip(): + return value.strip() + return "" + + +def _legacy_project_id(cwd: str, root: str, raw_remote: str, identity: str) -> str: + """Use the repository namespace created by the previous Mem0 plugin.""" + configured = _scope_value(os.environ.get("MEM0_PROJECT_ID")) + if configured: + return configured + + mapped = _scope_value(_legacy_project_map(cwd, root, raw_remote)) + if mapped: + return mapped + + remote = raw_remote or ("" if identity.startswith("local:") else identity) + remote = remote.strip().removesuffix(".git") + for prefix in ("https://", "http://", "ssh://", "git://"): + if remote.startswith(prefix): + remote = remote[len(prefix) :] + break + else: + remote = re.sub(r"^git@", "", remote) + parts = [part for part in remote.replace(":", "/", 1).split("/") if part] + if len(parts) >= 2: + return f"{parts[-2]}-{parts[-1]}".replace("/", "-").replace(":", "-") + if parts: + return parts[-1].replace("/", "-").replace(":", "-") + return os.path.basename(root or cwd) or "unknown" + + +@dataclass(frozen=True) +class RepoContext: + cwd: str + root: str + identity: str + app_id: str + branch: str + head_sha: str + project_id: str = "" + directory: str = "" + + +def _project_id(root: str, identity: str, app_id: str) -> str: + """The shared namespace: includes a host hash so repos with the same owner/name on different hosts stay apart.""" + if not identity.startswith("local:"): + return f"{app_id}-{hashlib.sha256(identity.encode()).hexdigest()[:10]}" + return f"local-{app_id}-{hashlib.sha256(root.encode()).hexdigest()[:10]}" + + +def _relative_directory(cwd: str, root: str) -> str: + relative = os.path.relpath(cwd, root) + return "" if relative == "." or relative.startswith("..") else relative.replace(os.sep, "/") + + +@dataclass(frozen=True) +class MemorySearchResult: + succeeded: bool + matched_count: int + already_shown_count: int + memories: list[dict[str, Any]] + + +@functools.lru_cache(maxsize=64) +def _resolve_repo_cached(cwd: str) -> RepoContext: + given_cwd = cwd + cwd = os.path.realpath(cwd) + given_root = _git(cwd, "rev-parse", "--show-toplevel") or given_cwd + root = os.path.realpath(given_root) + raw_remote = _git(root, "config", "--get", "remote.origin.url") + remote = _normalize_remote(raw_remote) + identity = remote or f"local:{root}" + app_id = _legacy_project_id(given_cwd, given_root, raw_remote, identity) + return RepoContext( + cwd=cwd, + root=root, + identity=identity, + app_id=app_id, + branch=_git(root, "branch", "--show-current") or "detached", + head_sha=_git(root, "rev-parse", "HEAD"), + project_id=_project_id(root, identity, app_id), + directory=_relative_directory(cwd, root), + ) + + +def resolve_repo(cwd: str | None) -> RepoContext: + return _resolve_repo_cached(os.path.abspath(cwd or os.getcwd())) + + +def api_key() -> str: + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return configured + try: + return (data_dir() / "api-key").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def cache_plugin_api_key() -> bool: + """Bridge host's hook-only sensitive config into plugin-owned storage.""" + configured = ( + os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if not configured: + return False + + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + path = directory / "api-key" + temporary = directory / f"api-key.{os.getpid()}.tmp" + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_TRUNC, + 0o600, + ) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + handle.write(configured) + os.replace(temporary, path) + os.chmod(path, 0o600) + finally: + try: + temporary.unlink() + except FileNotFoundError: + pass + return True + + +def clear_stale_api_key_cache() -> bool: + """Drop the cached key file once every configured key source is gone.""" + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return False + path = data_dir() / "api-key" + if not path.exists(): + return False + try: + path.unlink() + except OSError: + return False + return True + + +def detached_process_kwargs(platform: str | None = None) -> dict: + """Keep a spawned worker alive after the coding agent exits, on POSIX and Windows.""" + if (platform or sys.platform) == "win32": + return { + "creationflags": subprocess.DETACHED_PROCESS + | subprocess.CREATE_NEW_PROCESS_GROUP + } + return {"start_new_session": True} + + +def _plugin_option(name: str, fallback: str = "") -> str: + return ( + os.environ.get(f"PLUGIN_OPTION_{name.upper()}") + or os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") + or os.environ.get(fallback) + or "" + ).strip() + + +def user_id() -> str: + return ( + _scope_value(_plugin_option("user_id", "MEM0_CODE_USER_ID")) + or _scope_value(os.environ.get("MEM0_USER_ID")) + or _scope_value(os.environ.get("MEM0_RESOLVED_USER_ID")) + or _scope_value(os.environ.get("USER")) + or _scope_value(os.environ.get("USERNAME")) + or "default" + ) + + +def data_dir() -> Path: + configured = ( + os.environ.get("MEM0_CODE_DATA_DIR") + or os.environ.get("MEM0_PLUGIN_DATA_DIR") + or os.environ.get("PLUGIN_DATA") + or os.environ.get("CLAUDE_PLUGIN_DATA") + ) + return ( + Path(configured).expanduser() if configured else Path.home() / ".mem0" / _harness_data_dir_name + ) + + +def _bool_option(name: str, fallback: str, default: bool = False) -> bool: + value = _plugin_option(name, fallback) + if not value: + return default + return value.lower() in {"1", "true", "yes", "on"} + + +def _int_option(name: str, fallback: str, default: int) -> int: + value = _plugin_option(name, fallback) + try: + return int(value) if value else default + except ValueError: + return default + + + +def _checkpoint_message(event: dict[str, Any]) -> str: + kind = event.get("kind") + payload = event.get("payload") or {} + if kind == "user_prompt": + return redact(payload.get("text", "")).strip() + if kind == "assistant_stop": + transcript_messages = payload.get("transcript_messages") or [] + if isinstance(transcript_messages, list): + text = "\n".join( + str(message.get("content") or "") + for message in transcript_messages + if isinstance(message, dict) and message.get("content") + ) + if text: + return text + return redact(payload.get("text", "")).strip() + if kind == "sidekick_stop": + return redact(payload.get("final_message", "")).strip() + return "" + + +def checkpoint_stats(events: list[dict[str, Any]]) -> tuple[int, int, int]: + """Return completed exchanges, messages, and source characters.""" + completed = sum(event.get("kind") == "assistant_stop" for event in events) + contents = [content for event in events if (content := _checkpoint_message(event))] + return completed, len(contents), sum(len(content) for content in contents) + + +def select_checkpoint_events( + events: list[dict[str, Any]], *, force: bool +) -> list[dict[str, Any]]: + """Select one ordered extraction block without splitting an exchange.""" + for index, event in enumerate(events): + if event.get("kind") != "assistant_stop": + continue + candidate = events[: index + 1] + completed, messages, source_chars = checkpoint_stats(candidate) + if ( + completed >= CHECKPOINT_EXCHANGES + or messages >= CHECKPOINT_MESSAGES + or source_chars >= CHECKPOINT_SOURCE_CHARS + ): + return candidate + return events if force else [] + + +class EvidenceStore: + def __init__(self, path: Path | None = None): + directory = data_dir() if path is None else path.parent + directory.mkdir(parents=True, exist_ok=True) + self.path = path or directory / "evidence.sqlite3" + try: + self._open() + except sqlite3.DatabaseError: + self._quarantine() + self._open() + + def _open(self) -> None: + self.conn = sqlite3.connect(self.path, timeout=10) + self.conn.row_factory = sqlite3.Row + try: + self.conn.execute("PRAGMA journal_mode=WAL") + self.conn.execute("PRAGMA busy_timeout=10000") + self._migrate() + except sqlite3.DatabaseError: + self.conn.close() + raise + + def _quarantine(self) -> None: + """Move an unreadable database aside so capture restarts cleanly.""" + stamp = int(time.time()) + for suffix in ("", "-wal", "-shm"): + source = Path(f"{self.path}{suffix}") + try: + source.replace(f"{self.path}.corrupt-{stamp}{suffix}") + except FileNotFoundError: + continue + except OSError: + try: + source.unlink() + except OSError: + pass + telemetry.record("db_quarantined") + + def close(self) -> None: + self.conn.close() + + def _migrate(self) -> None: + self.conn.executescript( + """ + CREATE TABLE IF NOT EXISTS events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + created_at TEXT NOT NULL, + kind TEXT NOT NULL, + payload_json TEXT NOT NULL, + flush_id TEXT + ); + CREATE INDEX IF NOT EXISTS events_session_idx + ON events(repo_id, session_id, flush_id, id); + + CREATE TABLE IF NOT EXISTS session_scopes ( + session_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + root TEXT NOT NULL, + branch TEXT NOT NULL, + head_sha TEXT NOT NULL, + created_at TEXT NOT NULL, + directory TEXT NOT NULL DEFAULT '' + ); + + CREATE TABLE IF NOT EXISTS flushes ( + packet_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + reason TEXT NOT NULL, + event_start INTEGER NOT NULL, + event_end INTEGER NOT NULL, + status TEXT NOT NULL, + episode_event_id TEXT, + semantic_event_id TEXT, + error TEXT, + attempts INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + CREATE TABLE IF NOT EXISTS retrievals ( + session_id TEXT NOT NULL, + repo_id TEXT NOT NULL, + memory_id TEXT NOT NULL, + injected_at TEXT NOT NULL, + rank INTEGER, + score REAL, + memory_text TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(session_id, repo_id, memory_id) + ); + + CREATE TABLE IF NOT EXISTS operations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + created_at TEXT NOT NULL, + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + operation TEXT NOT NULL, + duration_ms REAL NOT NULL, + success INTEGER NOT NULL, + item_count INTEGER NOT NULL DEFAULT 0, + request_chars INTEGER NOT NULL DEFAULT 0, + response_chars INTEGER NOT NULL DEFAULT 0, + error TEXT + ); + + CREATE TABLE IF NOT EXISTS sidekick_runs ( + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + agent_type TEXT NOT NULL, + started_at TEXT NOT NULL, + stopped_at TEXT, + transcript_path TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + final_message TEXT, + PRIMARY KEY(repo_id, session_id, agent_id) + ); + CREATE INDEX IF NOT EXISTS sidekick_runs_repo_idx + ON sidekick_runs(repo_id, started_at); + + CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + """ + ) + # Remove the pre-0.1.1 no-tools snapshot implementation. The real coding + # sidekick is a native Claude Code agent and stores no state in this DB. + self.conn.executescript( + """ + DROP TABLE IF EXISTS sidekick_calls; + DROP TABLE IF EXISTS sidekick_snapshots; + DROP TABLE IF EXISTS sidekick_state; + DROP TABLE IF EXISTS sidekick_packets; + """ + ) + self._ensure_column("retrievals", "rank", "INTEGER") + self._ensure_column("retrievals", "score", "REAL") + self._ensure_column("retrievals", "memory_text", "TEXT") + self._ensure_column("retrievals", "context_chars", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("flushes", "attempts", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("session_scopes", "directory", "TEXT NOT NULL DEFAULT ''") + self.conn.commit() + + def _ensure_column(self, table: str, column: str, declaration: str) -> None: + columns = { + str(row["name"]) + for row in self.conn.execute(f"PRAGMA table_info({table})").fetchall() + } + if column not in columns: + self.conn.execute(f"ALTER TABLE {table} ADD COLUMN {column} {declaration}") + + def record_event( + self, + repo: RepoContext, + session_id: str, + kind: str, + payload: dict[str, Any], + ) -> int: + cursor = self.conn.execute( + """INSERT INTO events + (repo_id, app_id, session_id, created_at, kind, payload_json) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + repo.app_id, + session_id, + utc_now(), + kind, + json.dumps(payload, ensure_ascii=False, sort_keys=True), + ), + ) + self.conn.commit() + return int(cursor.lastrowid) + + def record_assistant_response(self, repo: RepoContext, session_id: str, message: str) -> None: + """Ignore repeated response hooks until another prompt or a different answer arrives.""" + with self.conn: + # Serialize the check and insert across concurrent Stop and SessionEnd hooks. + self.conn.execute("BEGIN IMMEDIATE") + previous = self.conn.execute( + """SELECT kind, payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind IN ('user_prompt', 'assistant_stop') + ORDER BY id DESC LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if ( + previous is not None + and previous["kind"] == "assistant_stop" + and json.loads(previous["payload_json"]).get("text") == message + ): + return + self.record_event(repo, session_id, "assistant_stop", {"text": message}) + + def repo_for_session(self, session_id: str, cwd: str | None) -> RepoContext: + """Keep one project scope for every hook in a coding-agent session.""" + current = resolve_repo(cwd) + if session_id == "unknown-session": + return current + + with self.conn: + self.conn.execute( + """INSERT OR IGNORE INTO session_scopes + (session_id, repo_id, app_id, root, branch, head_sha, created_at, directory) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + current.identity, + current.app_id, + current.root, + current.branch, + current.head_sha, + utc_now(), + current.directory, + ), + ) + scope = self.conn.execute( + "SELECT * FROM session_scopes WHERE session_id = ?", (session_id,) + ).fetchone() + same_git_repo = ( + current.identity == scope["repo_id"] and bool(current.head_sha) + ) + pinned = current if same_git_repo else resolve_repo(str(scope["root"])) + return RepoContext( + cwd=current.cwd, + root=pinned.root, + identity=str(scope["repo_id"]), + app_id=str(scope["app_id"]), + branch=pinned.branch, + head_sha=pinned.head_sha, + project_id=pinned.project_id, + directory=str(scope["directory"] or ""), + ) + + def prepare_flush( + self, repo: RepoContext, session_id: str, reason: str + ) -> tuple[str, list[dict[str, Any]]] | None: + with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + existing = self.conn.execute( + """SELECT * FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') + ORDER BY created_at LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: + self.conn.execute( + "UPDATE flushes SET status = 'gave-up', updated_at = ? WHERE packet_id = ?", + (utc_now(), existing["packet_id"]), + ) + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status="gave-up", + success=False, + attempts=int(existing["attempts"] or 0), + ) + existing = None + if existing: + if reason != "periodic" and existing["reason"] == "periodic": + self.conn.execute( + "UPDATE flushes SET reason = ?, updated_at = ? WHERE packet_id = ?", + (reason, utc_now(), existing["packet_id"]), + ) + existing_rows = self.conn.execute( + "SELECT * FROM events WHERE flush_id = ? ORDER BY id", + (existing["packet_id"],), + ).fetchall() + if existing_rows: + return str(existing["packet_id"]), [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in existing_rows + ] + + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo.identity, session_id), + ).fetchall() + if not rows: + return None + + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + events = select_checkpoint_events(events, force=reason != "periodic") + if not events: + return None + event_start, event_end = events[0]["id"], events[-1]["id"] + packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" + packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] + now = utc_now() + + self.conn.execute( + """INSERT OR IGNORE INTO flushes + (packet_id, repo_id, app_id, session_id, reason, event_start, + event_end, status, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, 'prepared', ?, ?)""", + ( + packet_id, + repo.identity, + repo.app_id, + session_id, + reason, + event_start, + event_end, + now, + now, + ), + ) + event_ids = [event["id"] for event in events] + placeholders = ", ".join("?" for _ in event_ids) + self.conn.execute( + f"UPDATE events SET flush_id = ? " + f"WHERE id IN ({placeholders}) AND flush_id IS NULL", + (packet_id, *event_ids), + ) + return packet_id, events + + def checkpoint_due(self, repo_id: str, session_id: str) -> bool: + if self.has_inflight_flush(repo_id, session_id): + return False + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo_id, session_id), + ).fetchall() + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + return bool(select_checkpoint_events(events, force=False)) + + def has_inflight_flush(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status IN ('prepared', 'semantic-queued') + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def flush_record(self, packet_id: str) -> dict[str, Any] | None: + row = self.conn.execute( + "SELECT * FROM flushes WHERE packet_id = ?", (packet_id,) + ).fetchone() + return dict(row) if row else None + + def has_unflushed_events(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def unflushed_starts_with_session_start( + self, repo_id: str, session_id: str + ) -> bool: + row = self.conn.execute( + """SELECT kind FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id LIMIT 1""", + (repo_id, session_id), + ).fetchone() + return bool(row and row["kind"] == "session_start") + + def update_flush(self, packet_id: str, **fields: Any) -> None: + allowed = {"status", "episode_event_id", "semantic_event_id", "error"} + updates = {key: value for key, value in fields.items() if key in allowed} + updates["updated_at"] = utc_now() + clause = ", ".join(f"{key} = ?" for key in updates) + failed = str(fields.get("status", "")) in { + "error", + "semantic-failed", + "semantic-timeout", + "semantic-missing", + } + if failed: + clause += ", attempts = attempts + 1" + with self.conn: + self.conn.execute( + f"UPDATE flushes SET {clause} WHERE packet_id = ?", + [*updates.values(), packet_id], + ) + + def unseen( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> list[dict[str, Any]]: + seen = { + row["memory_id"] + for row in self.conn.execute( + "SELECT memory_id FROM retrievals WHERE session_id = ? AND repo_id = ?", + (session_id, repo_id), + ) + } + return [memory for memory in memories if str(memory.get("id", "")) not in seen] + + def mark_injected( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> None: + now = utc_now() + with self.conn: + for rank, memory in enumerate(memories, start=1): + memory_id = str(memory.get("id", "")) + if memory_id: + memory_text = bounded( + memory.get("memory") or memory.get("text") or "", + 4000, + ) + try: + score = float(memory["score"]) + except (KeyError, TypeError, ValueError): + score = None + self.conn.execute( + """INSERT OR IGNORE INTO retrievals + (session_id, repo_id, memory_id, injected_at, rank, + score, memory_text, context_chars) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + repo_id, + memory_id, + now, + rank, + score, + memory_text, + len(memory_text), + ), + ) + + def injected_memories( + self, session_id: str, repo_id: str + ) -> list[dict[str, Any]]: + """Return the exact memories already supplied to the main conversation.""" + rows = self.conn.execute( + """SELECT memory_id, rank, score, memory_text + FROM retrievals + WHERE session_id = ? AND repo_id = ? + ORDER BY COALESCE(rank, 2147483647), injected_at, memory_id""", + (session_id, repo_id), + ).fetchall() + return [ + { + "id": row["memory_id"], + "memory": row["memory_text"], + "score": row["score"], + } + for row in rows + if row["memory_text"] + ] + + def start_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + context_chars: int, + ) -> bool: + """Record one native sidekick instance and whether context was first sent.""" + with self.conn: + cursor = self.conn.execute( + """INSERT OR IGNORE INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + context_chars) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + utc_now(), + context_chars, + ), + ) + return int(cursor.rowcount) > 0 + + def stop_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + transcript_path: str, + final_message: str, + ) -> str: + now = utc_now() + self.conn.execute("BEGIN IMMEDIATE") + try: + if not agent_id: + rows = self.conn.execute( + """SELECT agent_id FROM sidekick_runs + WHERE repo_id = ? AND session_id = ? AND agent_type = ? AND stopped_at IS NULL + LIMIT 2""", + (repo.identity, session_id, agent_type), + ).fetchall() + # Without a host ID, overlapping runs cannot be correlated reliably. + agent_id = rows[0]["agent_id"] if len(rows) == 1 else f"unknown-agent-{time.time_ns()}" + self.conn.execute( + """INSERT INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + stopped_at, transcript_path, final_message) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(repo_id, session_id, agent_id) DO UPDATE SET + stopped_at = excluded.stopped_at, + transcript_path = excluded.transcript_path, + final_message = excluded.final_message""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + now, + now, + bounded(transcript_path, 2000), + redact(final_message).strip(), + ), + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + return agent_id + + def operation( + self, + repo: RepoContext, + session_id: str, + operation: str, + duration_ms: float, + success: bool, + *, + item_count: int = 0, + request_chars: int = 0, + response_chars: int = 0, + error: str = "", + ) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO operations + (created_at, repo_id, session_id, operation, duration_ms, + success, item_count, request_chars, response_chars, error) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + ( + utc_now(), + repo.identity, + session_id, + operation, + duration_ms, + int(success), + item_count, + request_chars, + response_chars, + bounded(error, 1000), + ), + ) + + def has_operation(self, repo_id: str, session_id: str, operation: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM operations + WHERE repo_id = ? AND session_id = ? AND operation = ? + LIMIT 1""", + (repo_id, session_id, operation), + ).fetchone() + return row is not None + + def has_event(self, repo_id: str, session_id: str, kind: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + return row is not None + + def latest_event_payload( + self, repo_id: str, session_id: str, kind: str + ) -> dict[str, Any]: + row = self.conn.execute( + """SELECT payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + ORDER BY id DESC LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + if not row: + return {} + try: + payload = json.loads(row["payload_json"]) + except json.JSONDecodeError: + return {} + return payload if isinstance(payload, dict) else {} + + def setting(self, key: str, default: str = "") -> str: + row = self.conn.execute( + "SELECT value FROM settings WHERE key = ?", (key,) + ).fetchone() + return str(row["value"]) if row else default + + def set_setting(self, key: str, value: str) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO settings(key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at""", + (key, value, utc_now()), + ) + + def is_paused(self) -> bool: + return self.setting("paused", "false").lower() in { + "1", + "true", + "yes", + "on", + } + + def forget_local_repo(self, repo_id: str) -> dict[str, int]: + tables = { + "events": "repo_id", + "session_scopes": "repo_id", + "flushes": "repo_id", + "retrievals": "repo_id", + "operations": "repo_id", + "sidekick_runs": "repo_id", + } + removed: dict[str, int] = {} + with self.conn: + for table, column in tables.items(): + cursor = self.conn.execute( + f"DELETE FROM {table} WHERE {column} = ?", (repo_id,) + ) + removed[table] = max(int(cursor.rowcount), 0) + return removed + + def status(self, repo_id: str) -> dict[str, Any]: + def count(table: str) -> int: + return int( + self.conn.execute( + f"SELECT COUNT(*) FROM {table} WHERE repo_id = ?", (repo_id,) + ).fetchone()[0] + ) + + last_operation = self.conn.execute( + """SELECT created_at, operation, duration_ms, success, item_count, error + FROM operations WHERE repo_id = ? ORDER BY id DESC LIMIT 1""", + (repo_id,), + ).fetchone() + last_sidekick = self.conn.execute( + """SELECT session_id, agent_id, agent_type, started_at, stopped_at, + context_chars + FROM sidekick_runs WHERE repo_id = ? + ORDER BY started_at DESC LIMIT 1""", + (repo_id,), + ).fetchone() + return { + "paused": self.is_paused(), + "events": count("events"), + "flushes": count("flushes"), + "retrievals": count("retrievals"), + "sidekick_runs": count("sidekick_runs"), + "last_operation": dict(last_operation) if last_operation else None, + "last_sidekick": dict(last_sidekick) if last_sidekick else None, + } + + +def _session_id(hook_input: dict[str, Any]) -> str: + return str(hook_input.get("session_id") or "unknown-session") + + +def record_session_start(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + store.record_event( + repo, + session_id, + "session_start", + { + "source": hook_input.get("source", "startup"), + "model": bounded(hook_input.get("model", ""), 200), + "branch": repo.branch, + "head_sha": repo.head_sha, + }, + ) + telemetry.record( + "session_start", + repo=repo, + session_id=session_id, + trigger=bounded(str(hook_input.get("source", "startup")), 60), + model=bounded(hook_input.get("model", ""), 200), + api_key_configured=bool(api_key()), + is_git_repo=not repo.identity.startswith("local:"), + ) + + +def record_user_prompt( + store: EvidenceStore, hook_input: dict[str, Any] +) -> tuple[RepoContext, str, str, bool]: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prompt = redact(hook_input.get("prompt", "")).strip() + is_first_prompt = not store.has_event(repo.identity, session_id, "user_prompt") + store.record_event(repo, session_id, "user_prompt", {"text": prompt}) + return repo, session_id, prompt, is_first_prompt + + +def _tool_result_preview(response: Any) -> str: + if isinstance(response, dict): + selected = {} + for key in ( + "stdout", + "stderr", + "output", + "content", + "error", + "filePath", + "success", + "interrupted", + ): + if key in response: + selected[key] = response[key] + response = selected or {"keys": sorted(response.keys())[:20]} + return bounded(response, MAX_RESULT_CHARS) + + +def tool_payload(hook_input: dict[str, Any], *, failed: bool | None = False) -> dict[str, Any]: + name = str(hook_input.get("tool_name") or "unknown") + tool_input = hook_input.get("tool_input") or {} + if not isinstance(tool_input, dict): + tool_input = {} + payload: dict[str, Any] = { + "tool": name, + "failed": failed, + "duration_ms": hook_input.get("duration_ms"), + "agent_role": "sidekick" if hook_input.get("agent_id") else "main", + } + if hook_input.get("agent_id"): + payload["agent_id"] = bounded(hook_input["agent_id"], 200) + if hook_input.get("agent_type"): + payload["agent_type"] = bounded(hook_input["agent_type"], 200) + + if name in {"Read", "Write", "Edit", "MultiEdit", "NotebookEdit"}: + path = tool_input.get("file_path") or tool_input.get("notebook_path") + if path: + payload["path"] = bounded(path, 1000) + if name in {"Write", "Edit", "MultiEdit", "NotebookEdit"}: + payload["mutation_chars"] = sum( + len(str(tool_input.get(key, ""))) + for key in ("content", "new_string", "new_source", "edits") + ) + elif name == "Bash" or "command" in tool_input: + command = bounded(tool_input.get("command", ""), MAX_COMMAND_CHARS) + payload["command"] = command + payload["command_kind"] = ( + "test" + if TEST_COMMAND_RE.search(command) + else "build" + if BUILD_COMMAND_RE.search(command) + else "shell" + ) + response = ( + hook_input.get("error") if failed else hook_input.get("tool_response") + ) + payload["result_preview"] = _tool_result_preview(response) + elif name in {"Grep", "Glob", "WebSearch", "WebFetch"}: + for key in ("pattern", "path", "query", "url"): + if tool_input.get(key): + payload[key] = bounded(tool_input[key], 1000) + else: + payload["input_keys"] = sorted(tool_input.keys())[:20] + if failed: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + + if failed and "error" not in payload: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + return payload + + +def record_tool( + store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool | None = False +) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + payload = tool_payload(hook_input, failed=failed) + if payload.get("path"): + payload["repo_path"] = _repo_relative_path(repo, str(payload["path"])) + store.record_event( + repo, + session_id, + "tool_failure" if failed else "tool_result", + payload, + ) + + +def record_sidekick_start( + store: EvidenceStore, hook_input: dict[str, Any], *, inject_context: bool = True +) -> str: + """Record a native sidekick and reuse the main turn's retrieved memories.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_id = bounded(hook_input.get("agent_id", "unknown-agent"), 200) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + context = combine_context( + format_context(store.injected_memories(session_id, repo.identity)) + ) + if not inject_context: + context = "" + first_start = store.start_sidekick( + repo, session_id, agent_id, agent_type, len(context) + ) + store.record_event( + repo, + session_id, + "sidekick_start", + { + "agent_id": agent_id, + "agent_type": agent_type, + "context_chars": len(context) if first_start else 0, + "worktree_root": bounded(repo.root, 2000), + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="start", + first_start=first_start, + context_chars=len(context) if first_start else 0, + ) + return context if first_start else "" + + +def record_sidekick_stop(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + agent_id = bounded(hook_input.get("agent_id", ""), 200) + final_message = redact(hook_input.get("last_assistant_message", "")).strip() + transcript_path = bounded(hook_input.get("agent_transcript_path", ""), 2000) + agent_id = store.stop_sidekick( + repo, + session_id, + agent_id, + agent_type, + transcript_path, + final_message, + ) + store.record_event( + repo, + session_id, + "sidekick_stop", + { + "agent_id": agent_id, + "agent_type": agent_type, + "transcript_path": transcript_path, + "final_message": final_message, + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="stop", + has_transcript=bool(transcript_path), + message_chars=len(final_message), + ) + + +def _ordered_unique(values: Iterable[str]) -> list[str]: + seen: set[str] = set() + result = [] + for value in values: + if value and value not in seen: + seen.add(value) + result.append(value) + return result + + +def _repo_relative_path(repo: RepoContext, value: str) -> str: + value = str(value or "").strip() + if not value: + return "" + try: + path = Path(value) + if path.is_absolute(): + try: + return path.resolve().relative_to(Path(repo.root).resolve()).as_posix() + except ValueError: + return "" + except (OSError, ValueError): + pass + return bounded(value, 1000) + + +def _render_command_lines(commands: list[dict[str, str]]) -> list[str]: + lines = [] + for command in commands: + line = f"- [{command['status']}/{command['kind']}] {command['command']}" + if command["result"]: + line += f" — {bounded(command['result'], 500).replace(chr(10), ' ')}" + lines.append(line) + return lines + + +def build_episode( + repo: RepoContext, + session_id: str, + packet_id: str, + events: list[dict[str, Any]], + *, + canonical_task: str = "", + task_outcome: str = "", +) -> tuple[str, dict[str, Any]]: + prompts = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "user_prompt" and e["payload"].get("text") + ] + assistant_conclusions = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "assistant_stop" and e["payload"].get("text") + ] + sidekick_outcomes = [ + redact(e["payload"].get("final_message", "")).strip() + for e in events + if e["kind"] == "sidekick_stop" and e["payload"].get("final_message") + ] + tools = [ + e["payload"] for e in events if e["kind"] in {"tool_result", "tool_failure"} + and e["payload"].get("agent_role", "main") == "main" + ] + read_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") == "Read" + ) + modified_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") in {"Write", "Edit", "MultiEdit", "NotebookEdit"} + ) + searches = [ + {key: t[key] for key in ("tool", "pattern", "path", "query", "url") if key in t} + for t in tools + if t.get("tool") in {"Grep", "Glob", "WebSearch", "WebFetch"} + ] + commands = [ + { + "command": t.get("command", ""), + "kind": t.get("command_kind", "shell"), + "status": "unknown" if t.get("failed", False) is None else "failed" if t.get("failed") else "succeeded", + "result": t.get("result_preview", ""), + } + for t in tools + if t.get("command") + ] + + task = bounded(canonical_task or (prompts[0] if prompts else ""), 4000) + conclusion = redact(assistant_conclusions[-1] if assistant_conclusions else "").strip() + outcome = bounded(task_outcome, 2000) + + extraction_messages: list[dict[str, str]] = [] + pending_user_messages: list[dict[str, str]] = [] + if task and not prompts: + pending_user_messages.append({"role": "user", "content": task}) + for event in events: + if event["kind"] == "user_prompt" and event["payload"].get("text"): + pending_user_messages.append( + { + "role": "user", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + elif event["kind"] == "assistant_stop": + transcript_messages = event["payload"].get("transcript_messages") or [] + if isinstance(transcript_messages, list) and transcript_messages: + transcript_users = { + redact(message.get("content") or "").strip() + for message in transcript_messages + if isinstance(message, dict) and message.get("role") == "user" + } + extraction_messages.extend( + message + for message in pending_user_messages + if message["content"].strip() not in transcript_users + ) + extraction_messages.extend( + { + "role": str(message.get("role") or ""), + "content": redact(message.get("content") or "").strip(), + } + for message in transcript_messages + if isinstance(message, dict) + and message.get("role") in {"user", "assistant"} + and message.get("content") + ) + else: + extraction_messages.extend(pending_user_messages) + if event["payload"].get("text"): + extraction_messages.append( + { + "role": "assistant", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + pending_user_messages = [] + elif event["kind"] == "sidekick_stop": + pass + extraction_messages.extend(pending_user_messages) + + structured = { + "packet_id": packet_id, + "repo": repo.identity, + "app_id": repo.app_id, + "session_id": session_id, + "branch": repo.branch, + "head_sha": repo.head_sha, + "task": task, + "task_outcome": outcome, + "assistant_conclusion": conclusion, + "user_messages": prompts, + "assistant_outcomes": assistant_conclusions, + "sidekick_outcomes": sidekick_outcomes, + "extraction_messages": extraction_messages, + "files_read": read_paths[:50], + "files_modified": modified_paths[:50], + "searches": searches[-30:], + "commands": commands[-30:], + } + lines = ["Coding-session episode"] + if task: + lines.extend(["", "Task:", task]) + if modified_paths: + lines.extend( + ["", "Files modified:", *[f"- {path}" for path in modified_paths[:50]]] + ) + if read_paths: + lines.extend(["", "Files read:", *[f"- {path}" for path in read_paths[:50]]]) + if commands: + lines.append("") + lines.append("Observed commands:") + lines.extend(_render_command_lines(commands[-30:])) + if searches: + lines.extend( + [ + "", + "Observed searches:", + *[ + f"- {json.dumps(item, ensure_ascii=False, sort_keys=True)}" + for item in searches[-20:] + ], + ] + ) + if conclusion: + lines.extend(["", "Agent conclusion:", conclusion]) + if outcome: + lines.extend(["", "Task outcome:", outcome]) + lines.extend( + [ + "", + f"Provenance: repo={repo.identity}; branch={repo.branch}; head={repo.head_sha}; packet={packet_id}", + ] + ) + content = "\n".join(lines) + return bounded(content, MAX_EPISODE_CHARS), structured + + +def build_semantic_evidence(structured: dict[str, Any]) -> str: + """Format changed paths for memory extraction. + + Test and build results remain in the local evidence store for diagnostics, + but are not useful repository knowledge by default and should not steer + memory extraction toward transient verification details. + """ + modified_paths = [ + bounded(path, 500) for path in structured.get("files_modified", [])[:20] + ] + commands = structured.get("commands") or [] + if not any(command.get("status") == "failed" for command in commands): + commands = [] + + if not modified_paths and not commands: + return "" + + lines = ["Additional repository details from this session"] + if modified_paths: + lines.extend( + [ + "", + "Changed paths:", + *[f"- {path}" for path in modified_paths], + ] + ) + if commands: + lines.extend(["", "Commands run in this session:", *_render_command_lines(commands)]) + return bounded("\n".join(lines), 8000) + + +def build_extraction_messages(structured: dict[str, Any]) -> list[dict[str, str]]: + """Build the session messages sent to Mem0 for memory extraction.""" + evidence = build_semantic_evidence(structured) + messages = [ + {"role": message["role"], "content": redact(message["content"]).strip()} + for message in structured.get("extraction_messages", []) + if message.get("role") in {"user", "assistant"} and message.get("content") + ] + if evidence: + for message in reversed(messages): + if message["role"] == "assistant": + message["content"] = f"{message['content']}\n\n{evidence}" + break + else: + messages.append({"role": "assistant", "content": evidence}) + + return messages + + +def _estimated_tokens(value: str) -> int: + """Conservatively estimate tokens without adding a tokenizer dependency.""" + ascii_chars = sum(ord(char) < 128 for char in value) + return math.ceil((ascii_chars * 0.4) + (len(value) - ascii_chars)) + + +def _message_tokens(messages: list[dict[str, str]]) -> int: + return _estimated_tokens(json.dumps(messages, ensure_ascii=False)) + + +def _is_agent_assignment(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent assignment (") + + +def _is_agent_response(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent response (") + + +def extraction_message_batches( + messages: list[dict[str, str]], + *, + max_tokens: int = MAX_EXTRACTION_INPUT_TOKENS, +) -> list[list[dict[str, str]]]: + """Keep exchanges together when possible; split oversized messages to enforce the request budget.""" + if not messages or _message_tokens(messages) <= max_tokens: + return [messages] + + exchanges: list[list[dict[str, str]]] = [] + exchange: list[dict[str, str]] = [] + for message in messages: + if message.get("role") == "user" and exchange: + exchanges.append(exchange) + exchange = [] + exchange.append(message) + if exchange: + exchanges.append(exchange) + + units: list[list[dict[str, str]]] = [] + for exchange in exchanges: + if _message_tokens(exchange) <= max_tokens: + units.append(exchange) + continue + index = 0 + while index < len(exchange): + message = exchange[index] + if ( + _is_agent_assignment(message) + and index + 1 < len(exchange) + and _is_agent_response(exchange[index + 1]) + ): + units.append(exchange[index : index + 2]) + index += 2 + else: + units.append([message]) + index += 1 + + bounded_units: list[list[dict[str, str]]] = [] + for unit in units: + if _message_tokens(unit) <= max_tokens: + bounded_units.append(unit) + continue + for message in unit: + remaining = message["content"] + while remaining: + low, high = 0, len(remaining) + while low < high: + middle = (low + high + 1) // 2 + if _message_tokens([{**message, "content": remaining[:middle]}]) <= max_tokens: + low = middle + else: + high = middle - 1 + if low == 0: + raise ValueError("Extraction token budget cannot fit a message") + bounded_units.append([{**message, "content": remaining[:low]}]) + remaining = remaining[low:] + + batches: list[list[dict[str, str]]] = [] + batch: list[dict[str, str]] = [] + for unit in bounded_units: + candidate = [*batch, *unit] + if batch and _message_tokens(candidate) > max_tokens: + batches.append(batch) + batch = list(unit) + else: + batch = candidate + if batch: + batches.append(batch) + return batches + + +def _request_json( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + raw = json.dumps(payload, ensure_ascii=False).encode() + request = urllib.request.Request( + url, + data=raw, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + parsed = json.loads(response_raw or b"{}") + return parsed, len(raw), len(response_raw) + + +def _request_json_with_network_retry( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + """Retry one transient connection failure without retrying API responses.""" + try: + return _request_json(url, key, payload, timeout) + except urllib.error.HTTPError: + raise + except (urllib.error.URLError, TimeoutError, OSError): + time.sleep(0.25) + return _request_json(url, key, payload, timeout) + + +def _get_json( + url: str, key: str, timeout: float +) -> tuple[dict[str, Any] | list[Any], int]: + request = urllib.request.Request( + url, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="GET", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + return json.loads(response_raw or b"{}"), len(response_raw) + + +def _event_id(response: dict[str, Any] | list[Any]) -> str: + return str(response.get("event_id", "")) if isinstance(response, dict) else "" + + +def _stored_event_ids(value: Any) -> list[str]: + raw = str(value or "") + if not raw.startswith("["): + return [] + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return [] + return [str(item or "") for item in parsed] if isinstance(parsed, list) else [] + + +def _result_count(response: dict[str, Any] | list[Any]) -> int: + if isinstance(response, dict): + results = response.get("results") + return len(results) if isinstance(results, list) else 0 + return len(response) if isinstance(response, list) else 0 + + +def touch_handoff_heartbeat() -> None: + """Mark the worker's handoff file alive so recovery does not relaunch it.""" + path = os.environ.get("MEM0_CODE_HANDOFF_PATH", "") + if not path: + return + try: + os.utime(path) + except OSError: + pass + + +def _wait_for_event(api_url: str, key: str, event_id: str) -> tuple[str, int, int]: + """Wait for extraction to finish before a later task can search the store.""" + if not event_id: + return "MISSING", 0, 0 + wait_seconds = float(os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120")) + poll_seconds = max(float(os.environ.get("MEM0_CODE_EVENT_POLL_SECONDS", "1")), 0.1) + deadline = time.monotonic() + wait_seconds + response_chars = 0 + while time.monotonic() < deadline: + touch_handoff_heartbeat() + try: + response, size = _get_json( + f"{api_url}/v1/event/{event_id}/", + key, + min(10, poll_seconds + 5), + ) + except urllib.error.HTTPError as exc: + if exc.code not in {408, 429} and exc.code < 500: + raise + time.sleep(poll_seconds) + continue + except (urllib.error.URLError, TimeoutError, OSError): + # The extraction job is durable server-side. A transient polling + # failure must not discard a job that may still complete normally. + time.sleep(poll_seconds) + continue + response_chars += size + status = ( + str(response.get("status", "UNKNOWN")) + if isinstance(response, dict) + else "UNKNOWN" + ) + if status in {"SUCCEEDED", "FAILED"}: + return status, response_chars, _result_count(response) + time.sleep(poll_seconds) + return "TIMEOUT", response_chars, 0 + + +def _record_flush( + repo: RepoContext, + session_id: str, + reason: str, + status: str, + elapsed: float, + **extra: Any, +) -> None: + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status=status, + success=status in {"semantic-succeeded", "nothing-to-flush"}, + duration_ms=round(elapsed, 2), + **extra, + ) + + +def flush_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + key = api_key() + if not key: + telemetry.record("flush", reason=reason, status="local-only", success=False) + return {"status": "local-only", "reason": "no-api-key"} + + session_id = _session_id(hook_input) + if session_id == "unknown-session": + telemetry.record("flush", reason=reason, status="no-session-id", success=False) + return {"status": "error", "reason": "no-session-id"} + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prepared = store.prepare_flush(repo, session_id, reason) + if prepared is None: + return {"status": "nothing-to-flush"} + packet_id, events = prepared + existing_flush = store.flush_record(packet_id) or {} + + _, structured = build_episode( + repo, + session_id, + packet_id, + events, + canonical_task=bounded(hook_input.get("task", ""), 4000), + task_outcome=bounded(hook_input.get("task_outcome", ""), 2000), + ) + + metadata = {"source": _harness_source_tag} + if repo.branch and repo.branch not in {"detached", "unknown"}: + metadata["branch"] = repo.branch + if repo.head_sha: + metadata["git_sha"] = repo.head_sha + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + add_url = f"{api_url}/v3/memories/add/" + + write_user = _scope_value(user_id()) + write_project = _scope_value(repo.project_id) + if not write_user or not _scope_value(repo.app_id) or not write_project: + telemetry.record("flush", reason=reason, status="unscoped", success=False) + return {"status": "error", "reason": "wildcard-scope"} + + body = { + "agent_id": write_project, + "user_id": write_user, + "app_id": repo.app_id, + "run_id": session_id, + "metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)}, + "agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS, + "custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS, + "custom_categories": CODING_MEMORY_CATEGORIES, + "infer": True, + } + + started = time.perf_counter() + try: + stored_events = _stored_event_ids(existing_flush.get("semantic_event_id")) + existing_event = ( + "" if stored_events else str(existing_flush.get("semantic_event_id") or "") + ) + if existing_event: + existing_status, existing_resp, existing_items = _wait_for_event( + api_url, key, existing_event + ) + if existing_status == "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=existing_event, + error="", + ) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + True, + item_count=existing_items, + response_chars=existing_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=existing_items, + resumed=True, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "semantic_status": existing_status, + "memory_count": existing_items, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + if existing_status == "TIMEOUT": + elapsed = (time.perf_counter() - started) * 1000 + error = "semantic extraction event timed out" + store.update_flush(packet_id, status="semantic-timeout", error=error) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + False, + response_chars=existing_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-timeout", + elapsed, + resumed=True, + error_kind="timeout", + ) + return { + "status": "semantic-timeout", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + + message_batches = [ + batch + for batch in extraction_message_batches( + build_extraction_messages(structured) + ) + if batch + ] + batches = [(body, messages) for messages in message_batches] + if not batches: + store.update_flush(packet_id, status="semantic-succeeded", error="") + return {"status": "nothing-to-flush", "packet_id": packet_id} + operation_name = "flush-retry" if stored_events else "flush" + semantic_events = stored_events[: len(batches)] + semantic_events += [""] * (len(batches) - len(semantic_events)) + semantic_req = 0 + semantic_resp = 0 + for index, (body, messages) in enumerate(batches): + if semantic_events[index]: + continue + semantic_response, request_chars, response_chars = _request_json( + add_url, + key, + {**body, "messages": messages}, + 15, + ) + semantic_events[index] = _event_id(semantic_response) + semantic_req += request_chars + semantic_resp += response_chars + store.update_flush( + packet_id, + status="semantic-queued", + semantic_event_id=json.dumps(semantic_events), + ) + + semantic_event = semantic_events[-1] + semantic_status = "SUCCEEDED" + event_resp = 0 + semantic_items = 0 + failed_event = semantic_event + for index, queued_event in enumerate(semantic_events): + status, response_chars, item_count = _wait_for_event( + api_url, key, queued_event + ) + event_resp += response_chars + semantic_items += item_count + if status != "SUCCEEDED": + semantic_status = status + failed_event = queued_event + if status in {"FAILED", "MISSING"}: + semantic_events[index] = "" + store.update_flush( + packet_id, semantic_event_id=json.dumps(semantic_events) + ) + break + if semantic_status != "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + error = f"semantic extraction event {semantic_status.lower()}" + store.update_flush( + packet_id, status=f"semantic-{semantic_status.lower()}", error=error + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + False, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + f"semantic-{semantic_status.lower()}", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + error_kind=telemetry.error_kind(error), + ) + return { + "status": f"semantic-{semantic_status.lower()}", + "packet_id": packet_id, + "semantic_event_id": failed_event, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=json.dumps(semantic_events), + error="", + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + True, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + request_chars=semantic_req, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": semantic_event, + "semantic_status": semantic_status, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + except Exception as exc: # hooks must fail open + elapsed = (time.perf_counter() - started) * 1000 + error = bounded(str(exc), 1000) + store.update_flush(packet_id, status="error", error=error) + store.operation(repo, session_id, "flush", elapsed, False, error=error) + _record_flush( + repo, + session_id, + reason, + "error", + elapsed, + error_kind=telemetry.error_kind(exc), + ) + return {"status": "error", "packet_id": packet_id, "error": error} + + +def checkpoint_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + """Run remote extraction at a durable boundary.""" + return flush_session(store, hook_input, reason) + + + +def search_memories( + store: EvidenceStore | None, + repo: RepoContext, + session_id: str | None, + query: str, + *, + top_k: int | None = None, + category: str | None = None, + scope: str | None = None, + run_id: str | None = None, + operation: str = "search", + timeout: float = 5, +) -> MemorySearchResult: + key = api_key() + if not key or not query.strip(): + return MemorySearchResult(False, 0, 0, []) + search_once = os.environ.get( + "MEM0_CODE_SEARCH_ONCE_PER_SESSION", "false" + ).lower() in { + "1", + "true", + "yes", + "on", + } + track_session = store is not None and bool(session_id) + if ( + search_once + and track_session + and store.has_operation(repo.identity, session_id, "search") + ): + return MemorySearchResult(False, 0, 0, []) + + result_limit = min( + max( + top_k + if top_k is not None + else _int_option("top_k", "MEM0_CODE_TOP_K", 3), + 1, + ), + 20, + ) + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ValueError(f"Unknown memory category: {category}") + user, project = _scope_value(user_id()), _scope_value(repo.project_id) + if not user or not project or not _scope_value(repo.app_id): + return MemorySearchResult(False, 0, 0, []) + filters = _search_filters(user, repo, resolve_search_scope(scope)) + if category: + filters = {"AND": [filters, {"categories": {"contains": category}}]} + if run_id: + filters = {"AND": [filters, {"run_id": run_id}]} + payload = { + "query": query, + "app_id": repo.app_id, + "filters": filters, + "top_k": result_limit, + "rerank": False, + "latest_only": True, + } + url = ( + os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + + "/v3/memories/search/" + ) + started = time.perf_counter() + try: + response, request_chars, response_chars = _request_json_with_network_retry( + url, key, payload, timeout + ) + memories = ( + response if isinstance(response, list) else response.get("results", []) + ) + memories = [ + memory + for memory in memories + if isinstance(memory, dict) + and (memory.get("metadata") or {}).get("record_kind") != "task_episode" + ][:result_limit] + if track_session: + returned_memories = store.unseen(session_id, repo.identity, memories) + store.mark_injected(session_id, repo.identity, returned_memories) + already_shown_count = len(memories) - len(returned_memories) + else: + returned_memories = memories + already_shown_count = 0 + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation( + repo, + session_id, + operation, + elapsed, + True, + item_count=len(returned_memories), + request_chars=request_chars, + response_chars=response_chars, + ) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=True, + duration_ms=round(elapsed, 2), + matched_count=len(memories), + returned_count=len(returned_memories), + already_shown_count=already_shown_count, + top_k=result_limit, + has_category=bool(category), + ) + return MemorySearchResult( + succeeded=True, + matched_count=len(memories), + already_shown_count=already_shown_count, + memories=returned_memories, + ) + except Exception as exc: + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation(repo, session_id, operation, elapsed, False, error=str(exc)) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=False, + duration_ms=round(elapsed, 2), + top_k=result_limit, + has_category=bool(category), + error_kind=telemetry.error_kind(exc), + ) + return MemorySearchResult(False, 0, 0, []) + + +def format_context( + memories: list[dict[str, Any]], + heading: str = "Relevant repository memories:", +) -> str: + if not memories: + return "" + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + lines = [heading] if heading else [] + for memory in memories: + text = re.sub( + r"\s+", + " ", + redact(memory.get("memory") or memory.get("text") or ""), + ) + text = text.strip() + if not text: + continue + branch = str((memory.get("metadata") or {}).get("branch") or "").strip() + branch_label = ( + f" [learnt on branch {branch}]" + if branch.casefold() not in {"", "main", "master", "unknown", "detached"} + else "" + ) + number = len(lines) if heading else len(lines) + 1 + entry = f"{number}. {text}{branch_label}" + candidate = "\n".join([*lines, entry]) + if len(candidate) <= limit: + lines.append(entry) + continue + if not lines or (heading and len(lines) == 1): + prefix = f"{number}. " + suffix = f"…{branch_label}" + available = ( + limit + - len("\n".join(lines)) + - (1 if lines else 0) + - len(prefix) + - len(suffix) + ) + if available > 0: + lines.append(prefix + text[:available].rstrip() + suffix) + break + minimum_lines = 2 if heading else 1 + return "\n".join(lines) if len(lines) >= minimum_lines else "" + + +def format_search_result(result: MemorySearchResult) -> str: + """Return only the text the coding agent needs from an explicit memory search.""" + if not result.succeeded: + return "Memory search failed." + if result.memories: + rendered = format_context(result.memories, heading="") + if rendered: + return rendered + return "No matching memories found." + + +def combine_context(*contexts: str) -> str: + """Combine memory sources under one hard budget without repeated lines.""" + seen: set[str] = set() + lines: list[str] = [] + for context in contexts: + for line in str(context or "").splitlines(): + normalized = re.sub(r"\s+", " ", line).strip().casefold() + if not normalized or normalized in seen: + continue + seen.add(normalized) + lines.append(line.rstrip()) + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + return bounded("\n".join(lines), limit) if lines else "" + + +def _scoped_memory_ids( + api_url: str, key: str, user: str, repo: RepoContext, include_project: bool +) -> list[str]: + """List this user's memory ids for this repository, plus the shared project memory when asked.""" + ids: list[str] = [] + seen: set[str] = set() + prefix = repo.app_id + _collect_memory_ids( + api_url, key, {"user_id": user}, ids, seen, + app_id_prefix=prefix, + ) + if include_project: + for project_id in _shared_project_ids(repo): + _collect_memory_ids( + api_url, key, {"agent_id": project_id}, ids, seen, + app_id_prefix=prefix, + ) + return ids + + +def _collect_memory_ids( + api_url: str, + key: str, + filters: dict[str, Any], + ids: list[str], + seen: set[str], + *, + app_id_prefix: str = "", +) -> None: + """Page through one list filter; the list endpoint returns nothing for an OR whose user branch has no memories.""" + payload = {"filters": filters} + for page in range(1, FORGET_MAX_PAGES + 1): + parsed, _, _ = _request_json( + f"{api_url}/v2/memories/?page={page}&page_size={FORGET_PAGE_SIZE}", + key, + payload, + 15, + ) + items = parsed.get("results") if isinstance(parsed, dict) else parsed + if not isinstance(items, list) or not items: + break + for item in items: + if not isinstance(item, dict): + continue + memory_id = str(item.get("id", "")) + if not memory_id or memory_id in seen: + continue + if app_id_prefix: + item_app_id = str(item.get("app_id") or "") + if item_app_id != app_id_prefix and not item_app_id.startswith(app_id_prefix + "/"): + continue + seen.add(memory_id) + ids.append(memory_id) + if len(items) < FORGET_PAGE_SIZE: + break + + +def _delete_memory(api_url: str, key: str, memory_id: str) -> bool: + request = urllib.request.Request( + f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/", + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="DELETE", + ) + try: + with urllib.request.urlopen(request, timeout=15): + return True + except Exception: + return False + + +def forget_remote_repo( + repo: RepoContext, *, include_project_memory: bool = False +) -> dict[str, Any]: + """Delete this user's memories for this repository; project memory is shared, so only on request.""" + key = api_key() + if not key: + telemetry.record("forget", repo=repo, success=False, error_kind="no-api-key") + return {"status": "error", "error": "Mem0 API key is not configured"} + user = _scope_value(user_id()) + if not user or not _scope_value(repo.app_id) or not _scope_value(repo.project_id): + telemetry.record("forget", repo=repo, success=False, error_kind="unscoped") + return { + "status": "error", + "error": "Refusing to forget: the user or repository scope is a wildcard", + } + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + try: + memory_ids = _scoped_memory_ids(api_url, key, user, repo, include_project_memory) + except Exception as exc: + telemetry.record( + "forget", repo=repo, success=False, error_kind=telemetry.error_kind(exc) + ) + return {"status": "error", "error": bounded(str(exc), 1000)} + deleted = sum(_delete_memory(api_url, key, memory_id) for memory_id in memory_ids) + failed = len(memory_ids) - deleted + telemetry.record("forget", repo=repo, success=not failed, item_count=deleted) + if failed: + return { + "status": "partial", + "deleted": deleted, + "failed": failed, + "error": f"{failed} of {len(memory_ids)} memories could not be deleted", + } + return {"status": "deleted", "deleted": deleted} + + +def _doctor_mem0_authentication(repo: RepoContext) -> dict[str, Any]: + """Verify the configured key with one read-only, repository-scoped search.""" + key = api_key() + if not key: + return {"ok": False, "detail": "API key missing"} + payload = { + "query": "Mem0 authentication check", + "filters": { + "AND": [ + {"user_id": user_id()}, + {"app_id": repo.app_id}, + ] + }, + "top_k": 1, + "threshold": 1.0, + "rerank": False, + } + url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + started = time.perf_counter() + try: + _request_json(f"{url}/v3/memories/search/", key, payload, 5) + except Exception as exc: + return {"ok": False, "detail": bounded(str(exc), 300)} + elapsed = (time.perf_counter() - started) * 1000 + return {"ok": True, "detail": f"connected ({elapsed:.0f} ms)"} + + +def _doctor_user_id() -> dict[str, Any]: + """Flag a configured user ID the plugin refuses, since the silent fallback surprises people.""" + configured = _plugin_option("user_id", "MEM0_CODE_USER_ID") or os.environ.get( + "MEM0_USER_ID", "" + ) + if configured and not _scope_value(configured): + return { + "ok": False, + "detail": f"configured user_id {configured!r} is a wildcard; using {user_id()!r}", + } + return {"ok": True, "detail": user_id()} + + +def doctor(cwd: str | None = None) -> dict[str, Any]: + repo = resolve_repo(cwd) + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + checks: dict[str, dict[str, Any]] = { + "python": { + "ok": tuple(sys.version_info[:2]) >= (3, 10), + "detail": f"{sys.version_info.major}.{sys.version_info.minor}", + }, + "data_directory": { + "ok": os.access(directory, os.W_OK), + "detail": str(directory), + }, + "mem0_api_key": { + "ok": bool(api_key()), + "detail": "configured" if api_key() else "missing", + }, + "repository": { + "ok": bool(repo.identity), + "detail": repo.identity, + }, + "user_id": _doctor_user_id(), + "mem0_authentication": _doctor_mem0_authentication(repo), + } + return { + "ok": all(bool(value["ok"]) for value in checks.values()), + "plugin_version": PLUGIN_VERSION, + "repo_id": repo.identity, + "app_id": repo.app_id, + "user_id": user_id(), + "checks": checks, + } diff --git a/integrations/codex-plugin/core/telemetry.py b/integrations/codex-plugin/core/telemetry.py new file mode 100644 index 000000000..249595475 --- /dev/null +++ b/integrations/codex-plugin/core/telemetry.py @@ -0,0 +1,397 @@ +#!/usr/bin/env python3 +"""Anonymous usage telemetry for Mem0 agent plugins. + +Hooks run on a 3-6 second budget and fire on every tool call, so recording never +touches the network: `record` appends one JSON line to a local spool and returns. +A detached `python3 telemetry.py` drains the spool in one batched PostHog request, +started once per session and again from the flush worker that is already detached. + +Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false. + +Never sends prompts, memory text, queries, file paths, repository names, or API +keys: only event names, durations, counts, coarse outcomes, and salted hashes. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import platform +import subprocess +import sys +import time +import urllib.error +import urllib.request +import uuid +from pathlib import Path +from typing import Any + +import memory_core + +_harness: str = "generic" +_source_tag: str = "MEM0_PLUGIN" +_PRIVATE_KEYS = { + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +} + + +def init(harness: str = "generic", source_tag: str = "") -> None: + global _harness, _source_tag + _harness = harness + _source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN" + +POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" +POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" +POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/" +EVENT_PREFIX = "code" +SPOOL_LIMIT_BYTES = 256 * 1024 +BATCH_SIZE = 100 +SEND_TIMEOUT = 5 +CLAIM_STALE_SECONDS = 120 +CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60 + + +def is_enabled() -> bool: + """Whether telemetry is switched on for this process.""" + return os.environ.get("MEM0_TELEMETRY", "true").strip().lower() not in { + "false", + "0", + "no", + "off", + } + + +def _digest(value: str, length: int = 16) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length] + + +def _safe_value(value: Any) -> Any: + if isinstance(value, str): + return memory_core.redact(value) + if isinstance(value, dict): + return { + key: _safe_value(item) + for key, item in value.items() + if "".join(character for character in str(key).lower() if character.isalnum()) + not in _PRIVATE_KEYS + } + if isinstance(value, (list, tuple)): + return [_safe_value(item) for item in value] + if value is None or isinstance(value, (bool, int, float)): + return value + return memory_core.redact(value) + + +def _spool_path() -> Path: + return memory_core.data_dir() / "telemetry.jsonl" + + +def _identity_path() -> Path: + return memory_core.data_dir() / "telemetry-identity.json" + + +def _read_identity() -> dict[str, str]: + try: + value = json.loads(_identity_path().read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + return value if isinstance(value, dict) else {} + + +def _write_identity(identity: dict[str, str]) -> None: + path = _identity_path() + temporary = path.with_suffix(f".{os.getpid()}.tmp") + try: + path.parent.mkdir(parents=True, exist_ok=True) + temporary.write_text(json.dumps(identity), encoding="utf-8") + temporary.replace(path) + except OSError: + try: + temporary.unlink() + except OSError: + pass + + +def anonymous_id(identity: dict[str, str] | None = None) -> str: + """Per-machine anonymous identifier, created and persisted on first use.""" + identity = _read_identity() if identity is None else identity + existing = identity.get("anonymous_id") + if existing: + return existing + created = f"code-anon-{uuid.uuid4().hex}" + identity["anonymous_id"] = created + _write_identity(identity) + return created + + +def is_first_run() -> bool: + """Whether this machine has never recorded a plugin event before.""" + return not _identity_path().exists() + + +def record( + event: str, + *, + repo: Any = None, + session_id: str | None = None, + **properties: Any, +) -> None: + """Append one event to the local spool. Never blocks and never raises.""" + if not is_enabled(): + return + try: + spool = _spool_path() + try: + if spool.stat().st_size > SPOOL_LIMIT_BYTES: + return + except OSError: + pass + properties = _safe_value(properties) + properties.update( + harness=_harness, + plugin_version=memory_core.PLUGIN_VERSION, + os=sys.platform, + python_version=platform.python_version(), + ) + if repo is not None: + properties["repo_hash"] = _digest(getattr(repo, "identity", "")) + if session_id: + properties["session_hash"] = _digest(session_id) + line = json.dumps( + { + "event": f"{EVENT_PREFIX}.{event}", + "timestamp": memory_core.utc_now(), + "properties": { + key: value for key, value in properties.items() if value is not None + }, + }, + separators=(",", ":"), + default=str, + ) + spool.parent.mkdir(parents=True, exist_ok=True) + with spool.open("a", encoding="utf-8") as handle: + handle.write(line + "\n") + except Exception: + pass + + +def error_kind(exc: BaseException | str) -> str: + """Coarse, content-free label for a failure, safe to send.""" + text = exc if isinstance(exc, str) else f"{type(exc).__name__}: {exc}" + lowered = text.lower() + if "timed out" in lowered or "timeout" in lowered: + return "timeout" + if "401" in lowered or "403" in lowered or "unauthor" in lowered or "forbidden" in lowered: + return "auth" + if "429" in lowered or "rate limit" in lowered: + return "rate-limited" + if any(code in lowered for code in ("500", "502", "503", "504")): + return "server-error" + if "400" in lowered or "422" in lowered: + return "bad-request" + if isinstance(exc, str): + return "other" + if isinstance(exc, urllib.error.URLError): + return "network" + return type(exc).__name__ + + +def spawn_flush() -> bool: + """Start the detached sender that drains the spool.""" + if not is_enabled(): + return False + try: + if not _spool_path().exists() and not any( + memory_core.data_dir().glob("telemetry-*.sending") + ): + return False + subprocess.Popen( + [sys.executable, str(Path(__file__).resolve())], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + close_fds=True, + **memory_core.detached_process_kwargs(), + ) + return True + except Exception: + return False + + +def _claim_spool() -> Path | None: + """Rename the spool aside so exactly one sender owns each batch.""" + directory = memory_core.data_dir() + claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending" + spool = _spool_path() + try: + spool.replace(claim) + return claim + except OSError: + pass + now = time.time() + for orphan in sorted(directory.glob("telemetry-*.sending")): + try: + age = now - orphan.stat().st_mtime + except OSError: + continue + if age > CLAIM_EXPIRY_SECONDS: + try: + orphan.unlink() + except OSError: + pass + continue + if age < CLAIM_STALE_SECONDS: + continue + try: + orphan.replace(claim) + return claim + except OSError: + continue + return None + + +def _resolve_email(key: str) -> str: + """Trade the API key for the account email so events join other Mem0 surfaces.""" + url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/" + request = urllib.request.Request( + url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json"} + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT) as response: + payload = json.loads(response.read().decode("utf-8")) + except Exception: + return "" + email = payload.get("user_email") if isinstance(payload, dict) else "" + return email if isinstance(email, str) else "" + + +def _post(payload: dict[str, Any], url: str) -> bool: + request = urllib.request.Request( + url, + data=json.dumps(payload, default=str).encode("utf-8"), + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT): + return True + except Exception: + return False + + +def resolve_distinct_id() -> tuple[str, str]: + """Return the PostHog distinct id and the anonymous id it replaced, if any.""" + identity = _read_identity() + email = identity.get("email", "") + if email: + return email, "" + key = memory_core.api_key() + if not key: + return anonymous_id(identity), "" + email = _resolve_email(key) + if not email: + return anonymous_id(identity), "" + previous = identity.get("anonymous_id", "") + identity["email"] = email + _write_identity(identity) + return email, previous + + +def flush() -> int: + """Drain claimed spools to PostHog and return the number of events sent.""" + if not is_enabled(): + return 0 + claim = _claim_spool() + if claim is None: + return 0 + try: + lines = claim.read_text(encoding="utf-8").splitlines() + except OSError: + return 0 + events = [] + for line in lines: + try: + value = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(value, dict) and value.get("event"): + events.append(value) + if not events: + try: + claim.unlink() + except OSError: + pass + return 0 + + distinct_id, aliased_anonymous_id = resolve_distinct_id() + if aliased_anonymous_id: + _post( + { + "api_key": POSTHOG_API_KEY, + "event": "$identify", + "distinct_id": distinct_id, + "properties": { + "$anon_distinct_id": aliased_anonymous_id, + "$lib": "posthog-python", + }, + }, + POSTHOG_CAPTURE_URL, + ) + + sent = 0 + for start in range(0, len(events), BATCH_SIZE): + batch = [ + { + "event": event["event"], + "distinct_id": distinct_id, + "timestamp": event.get("timestamp"), + "properties": { + "source": _source_tag, + "language": "python", + "$process_person_profile": False, + "$lib": "posthog-python", + **(event.get("properties") or {}), + }, + } + for event in events[start : start + BATCH_SIZE] + ] + if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL): + return sent + sent += len(batch) + try: + claim.unlink() + except OSError: + pass + return sent + + +def main() -> int: + flush() + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception: + raise SystemExit(0) diff --git a/integrations/codex-plugin/hooks/adapter.py b/integrations/codex-plugin/hooks/adapter.py new file mode 100644 index 000000000..1fb6b0b3a --- /dev/null +++ b/integrations/codex-plugin/hooks/adapter.py @@ -0,0 +1,64 @@ +#!/usr/bin/env python3 +"""Run Codex hooks through the shared Mem0 runtime.""" + +from __future__ import annotations + +import sys +from pathlib import Path + +HERE = Path(__file__).resolve() +BUNDLED_CORE = HERE.parent.parent / "core" +CORE = BUNDLED_CORE if BUNDLED_CORE.is_dir() else HERE.parents[2] / "core" / "python" +sys.path.insert(0, str(CORE)) + +import hook_runner # noqa: E402 +import telemetry # noqa: E402 +from memory_core import ( # noqa: E402 + configure_harness, + record_sidekick_start, + record_sidekick_stop, + record_tool, +) + +configure_harness("codex", data_dir_name="codex-plugin", source_tag="codex_plugin") +telemetry.init(harness="codex", source_tag="CODEX_PLUGIN") + + +def _sidekick_start(store, hook_input): + context = record_sidekick_start(store, hook_input) + if context: + return { + "hookSpecificOutput": { + "hookEventName": "SubagentStart", + "additionalContext": context, + } + } + + +def _sidekick_stop(store, hook_input): + record_sidekick_stop(store, hook_input) + + +def _post_tool(store, payload): + response = payload.get("tool_response") + failed = None + if isinstance(response, dict): + exit_code = response.get("exit_code") + if isinstance(exit_code, int) and not isinstance(exit_code, bool): + failed = exit_code != 0 + elif isinstance(response.get("isError"), bool): + failed = response["isError"] + elif isinstance(response.get("success"), bool): + failed = not response["success"] + if failed: + payload = {**payload, "error": response} + record_tool(store, payload, failed=failed) + + +if __name__ == "__main__": + if len(sys.argv) > 1 and sys.argv[1] == "post-tool": + sys.argv[1] = "codex-post-tool" + hook_runner.entry_point( + extra_actions={"sidekick-start": _sidekick_start, "sidekick-stop": _sidekick_stop, "codex-post-tool": _post_tool}, + automatic_flush_reasons={"session-end", "pre-compact"}, + ) diff --git a/integrations/codex-plugin/hooks/hooks.json b/integrations/codex-plugin/hooks/hooks.json new file mode 100644 index 000000000..44c3732a1 --- /dev/null +++ b/integrations/codex-plugin/hooks/hooks.json @@ -0,0 +1,63 @@ +{ + "description": "Create memories from coding sessions and recall them in later work.", + "hooks": { + "SessionStart": [ + { + "matcher": "startup|resume|clear|compact", + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" session-start --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ], + "UserPromptSubmit": [ + { + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" user-prompt --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ], + "PostToolUse": [ + { + "matcher": ".*", + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" post-tool --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ], + "SubagentStart": [ + { + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" sidekick-start --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ], + "SubagentStop": [ + { + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" sidekick-stop --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ], + "Stop": [ + { + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" stop --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ], + "PreCompact": [ + { + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" flush --reason pre-compact --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ], + "SessionEnd": [ + { + "hooks": [ + { "type": "command", "command": "python3 \"${PLUGIN_ROOT}/hooks/adapter.py\" flush --reason session-end --plugin-data-dir \"${PLUGIN_DATA}\"", "timeout": 3 } + ] + } + ] + } +} diff --git a/integrations/codex-plugin/plugin-build.json b/integrations/codex-plugin/plugin-build.json new file mode 100644 index 000000000..8d4c51c07 --- /dev/null +++ b/integrations/codex-plugin/plugin-build.json @@ -0,0 +1,15 @@ +{ + "id": "mem0", + "version": "0.3.1", + "homepage": "https://docs.mem0.ai/integrations/codex", + "native": { + "pluginRoot": "${PLUGIN_ROOT}", + "pluginData": "${PLUGIN_DATA}", + "files": { + ".codex-plugin/plugin.json": ".codex-plugin/plugin.json", + "hooks/hooks.json": "hooks/hooks.json", + "hooks/adapter.py": "hooks/adapter.py", + ".mcp.json": ".mcp.json" + } + } +} diff --git a/integrations/codex-plugin/skills/forget/SKILL.md b/integrations/codex-plugin/skills/forget/SKILL.md new file mode 100644 index 000000000..09b285cac --- /dev/null +++ b/integrations/codex-plugin/skills/forget/SKILL.md @@ -0,0 +1,26 @@ +--- +name: forget +description: Delete the Mem0 memories stored for this repository and this user. Use when the user asks to forget, clear, wipe, or delete memories. +disable-model-invocation: true +--- + +# Forget this repository's memories + +This permanently deletes remote memories. Before running anything, tell the +user exactly what will be deleted: their own memories for this repository +only. The repository's project memory is shared by everyone who works in it, +so it stays unless the user explicitly asks to delete that too. + +After the user confirms, run: + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "codex" --plugin-data-dir "${PLUGIN_DATA}" forget --remote --yes +``` + +If the user also asked to delete the repository's shared project memory, add +`--include-project-memory` and say that this removes it for every teammate. + +Report what the command output says was deleted. If the user only wants local +data cleared (evidence log, pending queue), run the same command without +`--remote`. Never pass `--yes` before the user has confirmed in this +conversation. diff --git a/integrations/codex-plugin/skills/pause/SKILL.md b/integrations/codex-plugin/skills/pause/SKILL.md new file mode 100644 index 000000000..d8a52f47b --- /dev/null +++ b/integrations/codex-plugin/skills/pause/SKILL.md @@ -0,0 +1,20 @@ +--- +name: pause +description: Pause Mem0 memory capture on this machine. Use when the user wants to stop memories being recorded, for example for private work or experiments. +disable-model-invocation: true +--- + +# Pause memory capture + +To pause (hooks stop capturing and sending session content; a minimal +anonymous telemetry ping still fires at session start unless +`MEM0_TELEMETRY=false`): + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "codex" --plugin-data-dir "${PLUGIN_DATA}" pause +``` + +Confirm the new state back to the user, and remind them that already-created +memories still exist and remain searchable. Pending unsent packets are held +while paused, not expired, and are delivered after resuming. To turn capture +back on, use `/mem0:resume`. diff --git a/integrations/codex-plugin/skills/remember/SKILL.md b/integrations/codex-plugin/skills/remember/SKILL.md new file mode 100644 index 000000000..7be5fc9be --- /dev/null +++ b/integrations/codex-plugin/skills/remember/SKILL.md @@ -0,0 +1,21 @@ +--- +name: remember +description: Acknowledge a "remember this" request and make sure it is captured well. Use when the user explicitly asks to remember, note, or save something for future sessions. +disable-model-invocation: true +--- + +# Remember something for future sessions + +Mem0 creates memories from the session automatically — there is no separate +write command. When the user asks to remember something: + +1. Restate the fact clearly and completely in your reply, in one or two + sentences, including any names, values, or paths it depends on. Your visible + reply is what memory extraction reads, so a precise restatement is what gets + remembered. +2. Tell the user it will be saved with this session's memories when the session + ends or compacts, and that it will surface in future sessions in this + repository (they can check later with /mem0:search). + +Do not invent a storage confirmation or a memory ID — creation happens in the +background after the session. diff --git a/integrations/codex-plugin/skills/resume/SKILL.md b/integrations/codex-plugin/skills/resume/SKILL.md new file mode 100644 index 000000000..59a097585 --- /dev/null +++ b/integrations/codex-plugin/skills/resume/SKILL.md @@ -0,0 +1,19 @@ +--- +name: resume +description: Resume Mem0 memory capture after it was paused with /mem0:pause. +disable-model-invocation: true +--- + +# Resume memory capture + +Resume memory capture for this machine. + +Run: + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "codex" --plugin-data-dir "${PLUGIN_DATA}" resume +``` + +Confirm to the user that capture is active again. New sessions record evidence and +create memories as normal; nothing that happened while paused is retroactively +captured. diff --git a/integrations/codex-plugin/skills/search/SKILL.md b/integrations/codex-plugin/skills/search/SKILL.md new file mode 100644 index 000000000..f6c27b19c --- /dev/null +++ b/integrations/codex-plugin/skills/search/SKILL.md @@ -0,0 +1,28 @@ +--- +name: search +description: Search memories from earlier Codex sessions in this repository. Use it when earlier work may already explain the code, error, decision, or command you need, so you can avoid repeating file reads, searches, or experiments. +argument-hint: "[question] [--top-k number] [--category category-name] [--scope repo|dir|mine] [--run-id session-id]" +disable-model-invocation: true +--- + +# Search memories + +Call `search_memories` with the user's question. Treat `--top-k`, `--category`, +`--scope`, and `--run-id` as tool arguments instead of including them in the +query. + +Omit `top_k` to use Mem0's configured default. Omit `category` to search every +category; a category is a best-effort label Mem0 assigned when it saved the +memory, so if a category search misses, repeat it without the category. Omit +`scope` to use the configured default, normally `repo`: this repository's +shared memory, which everyone who works in it contributes to, plus your own +preferences. + +Pass `scope` when the question needs something else: `dir` to narrow the +shared memory to the directory you are working in (a package inside a +monorepo), `mine` for your own preferences alone. + +Pass `run_id` with any scope to retrieve memories saved in a specific coding-agent +session. Omit `run_id` to search across sessions. It filters the memories returned; +it does not identify the session making the search request. Use a known session ID, +never invent one. Return the tool's result directly. diff --git a/integrations/codex-plugin/skills/status/SKILL.md b/integrations/codex-plugin/skills/status/SKILL.md new file mode 100644 index 000000000..4f67cc829 --- /dev/null +++ b/integrations/codex-plugin/skills/status/SKILL.md @@ -0,0 +1,23 @@ +--- +name: status +description: Show whether Mem0 memory is working in this repository, covering configuration, capture state, pending flushes, and whether the Mem0 API key is valid. Use when the user asks whether memory is on, why a memory is missing, or anything looks broken. +disable-model-invocation: false +--- + +# Memory status + +Run both commands and report the combined result in plain language: + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "codex" --plugin-data-dir "${PLUGIN_DATA}" status --json +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "codex" --plugin-data-dir "${PLUGIN_DATA}" doctor +``` + +Summarize, using only fields the JSON actually reports: whether capture is +active or paused, the user ID and repository scope (`repo_id`), whether an +API key is configured, the event/flush/retrieval counts (`flushes` is the +number of completed flushes, not a pending count), and the doctor check +results. If doctor reports an authentication failure (401 / invalid key), say +clearly that the Mem0 API key is invalid or expired and that memories are NOT +being created. Never report an auth failure as "no memories found". Suggest +reinstalling with `--config api_key=...` in that case. diff --git a/integrations/codex-plugin/tests/test_codex.py b/integrations/codex-plugin/tests/test_codex.py new file mode 100644 index 000000000..4254c1f13 --- /dev/null +++ b/integrations/codex-plugin/tests/test_codex.py @@ -0,0 +1,107 @@ +from __future__ import annotations + +import json +import sqlite3 +import subprocess +import sys +from pathlib import Path + +HOST = Path(__file__).resolve().parents[1] +CORE_ROOT = HOST.parent / "agent-plugin-core" +sys.path.insert(0, str(CORE_ROOT)) + +from build.build import build # noqa: E402 + + +def test_codex_hooks_use_native_events_and_plugin_paths() -> None: + hooks = json.loads((HOST / "hooks" / "hooks.json").read_text(encoding="utf-8"))["hooks"] + + assert set(hooks) == { + "SessionStart", + "UserPromptSubmit", + "PostToolUse", + "SubagentStart", + "SubagentStop", + "Stop", + "PreCompact", + "SessionEnd", + } + commands = [hook["command"] for groups in hooks.values() for group in groups for hook in group["hooks"]] + assert "matcher" not in hooks["SubagentStart"][0] + assert "matcher" not in hooks["SubagentStop"][0] + assert all("${PLUGIN_ROOT}/hooks/adapter.py" in command for command in commands) + assert any("flush --reason pre-compact" in command for command in commands) + assert any("flush --reason session-end" in command for command in commands) + assert all(hook.get("timeout", 0) <= 3 for groups in hooks.values() for group in groups for hook in group["hooks"]) + + +def test_codex_sidekick_records_lifecycle(tmp_path: Path) -> None: + adapter = HOST / "hooks" / "adapter.py" + payload = { + "session_id": "session-1", + "cwd": str(tmp_path), + "agent_id": "agent-1", + "agent_type": "default", + } + + for action, extra in ( + ("sidekick-start", {}), + ("sidekick-stop", {"last_assistant_message": "SIDEKICK_OK"}), + ): + result = subprocess.run( + [sys.executable, str(adapter), action, "--plugin-data-dir", str(tmp_path / "data")], + input=json.dumps({**payload, **extra}), + text=True, + capture_output=True, + check=False, + ) + assert result.returncode == 0, result.stderr + + with sqlite3.connect(tmp_path / "data" / "evidence.sqlite3") as connection: + row = connection.execute( + "SELECT agent_id, stopped_at, final_message FROM sidekick_runs" + ).fetchone() + + assert row is not None + assert row[0] == "agent-1" + assert row[1] + assert row[2] == "SIDEKICK_OK" + + +def test_codex_mcp_uses_host_relative_paths() -> None: + manifest = json.loads((HOST / ".codex-plugin" / "plugin.json").read_text(encoding="utf-8")) + assert manifest["mcpServers"] == "./.mcp.json" + + server = json.loads((HOST / ".mcp.json").read_text(encoding="utf-8"))["mcpServers"]["mem0"] + assert server["command"] == "python3" + assert server["args"] == ["./core/mcp_server.py"] + assert server["cwd"] == "." + assert set(server["env_vars"]) >= {"MEM0_API_KEY", "MEM0_CODE_USER_ID"} + assert "PLUGIN_ROOT" not in json.dumps(server) + + +def test_native_codex_bundle_is_self_contained(tmp_path: Path) -> None: + root = build("codex", "native", tmp_path / "codex") + + manifest = json.loads((root / ".codex-plugin" / "plugin.json").read_text(encoding="utf-8")) + assert manifest["hooks"] == "./hooks/hooks.json" + assert manifest["skills"] == "./skills/" + assert (root / "hooks" / "adapter.py").is_file() + assert (root / "core" / "mcp_server.py").is_file() + assert (root / ".mcp.json").is_file() + assert not (root / "agents").exists() + assert not any(path.is_symlink() for path in root.rglob("*")) + + +def test_codex_post_tool_keeps_failures_and_unknown_outcomes(tmp_path): + for response, failed in [("raw command output", None), ({"exit_code": 1, "stderr": "failed"}, True), ({"isError": False}, False)]: + result = subprocess.run( + [sys.executable, str(HOST / "hooks" / "adapter.py"), "post-tool", "--plugin-data-dir", str(tmp_path / "data")], + input=json.dumps({"session_id": "s1", "cwd": str(tmp_path), "tool_name": "Bash", + "tool_input": {"command": "test command"}, "tool_response": response}), + text=True, capture_output=True, check=False, + ) + assert result.returncode == 0, result.stderr + with sqlite3.connect(tmp_path / "data" / "evidence.sqlite3") as connection: + row = connection.execute("SELECT payload_json FROM events ORDER BY id DESC LIMIT 1").fetchone() + assert json.loads(row[0])["failed"] is failed diff --git a/integrations/cursor-plugin/.cursor-plugin/plugin.json b/integrations/cursor-plugin/.cursor-plugin/plugin.json new file mode 100644 index 000000000..fe9594a12 --- /dev/null +++ b/integrations/cursor-plugin/.cursor-plugin/plugin.json @@ -0,0 +1,25 @@ +{ + "name": "mem0", + "version": "0.3.1", + "description": "Cross-session memory and token savings for coding agents.", + "author": { "name": "Mem0", "email": "support@mem0.ai" }, + "homepage": "https://docs.mem0.ai/integrations/cursor", + "repository": "https://github.com/mem0ai/mem0", + "license": "Apache-2.0", + "keywords": ["memory", "coding-agents", "continual-learning", "token-efficiency"], + "skills": "./skills/", + "agents": "./agents/", + "hooks": "./hooks/hooks.json", + "mcpServers": "./mcp.json", + "variables": { + "type": "object", + "properties": { + "api_key": { "type": "string", "title": "Mem0 API Key", "description": "Mem0 Platform API key used to create and search memories." }, + "user_id": { "type": "string", "title": "Memory user ID", "description": "Optional stable ID shared across machines.", "default": "" }, + "top_k": { "type": "number", "title": "Manual search results", "default": 3, "minimum": 1, "maximum": 20 }, + "max_context_chars": { "type": "number", "title": "Maximum retrieved context", "default": 4000, "minimum": 1000, "maximum": 10000 }, + "search_scope": { "type": "string", "title": "Default search scope", "default": "repo", "enum": ["repo", "dir", "mine"] } + }, + "required": ["api_key"] + } +} diff --git a/integrations/cursor-plugin/agents/sidekick.md b/integrations/cursor-plugin/agents/sidekick.md new file mode 100644 index 000000000..31fa6e945 --- /dev/null +++ b/integrations/cursor-plugin/agents/sidekick.md @@ -0,0 +1,16 @@ +--- +name: sidekick +description: Coding agent with an isolated context. Delegate focused investigation, implementation, testing, debugging, or review work to it, then review its result. +model: inherit +--- + +You are Mem0's Cursor sidekick. Complete only the work the parent agent assigns. +Return a tested result the parent can review without repeating your investigation. + +Always call `search_memories` before work that can depend on earlier decisions, +repository history, or user preferences. Inspect the relevant code and repository +rules, make requested edits, and run the smallest decisive checks. + +Do not make adjacent improvements. Ask the parent one concise question only when +a material decision or unsafe ambiguity blocks progress. Otherwise proceed and +report the outcome, changed files, validation, and remaining risk. diff --git a/integrations/cursor-plugin/core/flush_worker.py b/integrations/cursor-plugin/core/flush_worker.py new file mode 100644 index 000000000..6f7b9ecb2 --- /dev/null +++ b/integrations/cursor-plugin/core/flush_worker.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Detached remote checkpoint worker. + +Claude Code may cancel SessionEnd hooks as a print-mode process exits. The hook +therefore persists its input first and launches this process in a new session. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + checkpoint_session, + configure_harness, + touch_handoff_heartbeat, +) + + +def main() -> int: + if len(sys.argv) != 2: + return 2 + handoff_path = Path(sys.argv[1]) + os.environ["MEM0_CODE_HANDOFF_PATH"] = str(handoff_path) + harness = os.environ.get("MEM0_PLUGIN_HARNESS") + if harness: + source_tag = os.environ.get("MEM0_PLUGIN_SOURCE_TAG", "") + configure_harness( + harness, + env_prefix=os.environ.get("MEM0_PLUGIN_ENV_PREFIX", ""), + data_dir_name=os.environ.get("MEM0_PLUGIN_DATA_DIR_NAME", ""), + source_tag=source_tag, + ) + telemetry.init(harness=harness, source_tag=source_tag.upper()) + completed = False + try: + payload = json.loads(handoff_path.read_text(encoding="utf-8")) + delay = float(payload.get("delay_seconds") or 0) + if delay > 0: + payload.pop("delay_seconds", None) + temporary = handoff_path.with_suffix(f".{os.getpid()}.tmp") + try: + temporary.write_text(json.dumps(payload), encoding="utf-8") + temporary.replace(handoff_path) + finally: + temporary.unlink(missing_ok=True) + time.sleep(delay) + if not handoff_path.exists(): + return 0 + hook_input = payload.get("hook_input") or {} + reason = str(payload.get("reason") or "checkpoint") + wait_for_inflight = bool(payload.get("wait_for_inflight")) + store = EvidenceStore() + try: + if wait_for_inflight: + session_id = str( + hook_input.get("session_id") or "unknown-session" + ) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + deadline = time.monotonic() + float( + os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120") + ) + while ( + store.has_inflight_flush(repo.identity, session_id) + and time.monotonic() < deadline + ): + touch_handoff_heartbeat() + time.sleep(0.25) + # Hooks capture the conversation before handoff; the worker only flushes it. + result = checkpoint_session(store, hook_input, reason) + print(json.dumps(result, sort_keys=True), flush=True) + completed = result.get("status") in { + "semantic-succeeded", + "explicitly-stored", + "nothing-to-flush", + } + finally: + store.close() + return 0 + finally: + telemetry.flush() + if completed: + try: + handoff_path.unlink() + except OSError: + pass + elif handoff_path.suffix == ".running": + try: + handoff_path.replace(handoff_path.with_suffix(".json")) + except OSError: + pass + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/cursor-plugin/core/hook_runner.py b/integrations/cursor-plugin/core/hook_runner.py new file mode 100644 index 000000000..2a2adb557 --- /dev/null +++ b/integrations/cursor-plugin/core/hook_runner.py @@ -0,0 +1,372 @@ +"""Shared hook orchestration for all Mem0 agent plugins.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import subprocess +import sys +import time +import uuid +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + _session_id, + api_key, + bounded, + cache_plugin_api_key, + checkpoint_session, + clear_stale_api_key_cache, + configure_harness, + data_dir, + detached_process_kwargs, + format_context, + harness_config, + record_session_start, + record_tool, + record_user_prompt, + redact, + search_memories, +) + +STALE_RUNNING_SECONDS = 300 +PENDING_EXPIRY_SECONDS = 7 * 24 * 60 * 60 +PENDING_LAUNCH_LIMIT = 5 +DEFAULT_IDLE_FLUSH_SECONDS = 300 + +_core_dir: Path = Path(__file__).resolve().parent + + +def read_hook_input() -> dict: + try: + value = json.load(sys.stdin) + return value if isinstance(value, dict) else {} + except (json.JSONDecodeError, OSError): + return {} + + +def default_record_stop(store: EvidenceStore, hook_input: dict): + """Record the assistant's response without transcript parsing.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + message = redact(hook_input.get("last_assistant_message", "")).strip() + if message: + store.record_assistant_response(repo, session_id, message) + return repo, session_id + + +def first_prompt_memory_output(store: EvidenceStore, hook_input: dict) -> dict: + """Search once before the agent handles the first prompt in a session.""" + repo, session_id, prompt, is_first_prompt = record_user_prompt(store, hook_input) + if not is_first_prompt: + return {} + try: + minimum_query_chars = int(os.environ.get("MEM0_CODE_MIN_QUERY_CHARS", "20")) + except ValueError: + minimum_query_chars = 20 + if len(prompt.strip()) < max(minimum_query_chars, 1): + return {} + result = search_memories( + store, repo, session_id, bounded(prompt, 6000), + top_k=5, operation="first-prompt-search", timeout=2, + ) + if not result.memories: + return {} + context = format_context( + result.memories, + "Mem0 found these relevant memories from earlier work in this repository:", + ) + telemetry.record( + "context_injected", + repo=repo, session_id=session_id, trigger="first-prompt", + memory_count=len(result.memories), context_chars=len(context), + prompt_chars=len(prompt), + ) + return { + "hookSpecificOutput": { + "hookEventName": "UserPromptSubmit", + "additionalContext": context, + }, + } + + +def _launch_handoff(handoff_path: Path) -> bool: + running_path = handoff_path.with_suffix(".running") + try: + handoff_path.replace(running_path) + except OSError: + return False + worker = _core_dir / "flush_worker.py" + log_path = data_dir() / "flush-worker.log" + log_handle = open(log_path, "a", encoding="utf-8") + harness = harness_config() + child_env = os.environ.copy() + child_env.update( + { + "MEM0_CODE_DATA_DIR": str(data_dir()), + "MEM0_PLUGIN_HARNESS": harness["name"], + "MEM0_PLUGIN_ENV_PREFIX": harness["env_prefix"], + "MEM0_PLUGIN_DATA_DIR_NAME": harness["data_dir_name"], + "MEM0_PLUGIN_SOURCE_TAG": harness["source_tag"], + } + ) + try: + subprocess.Popen( + [sys.executable, str(worker), str(running_path)], + stdin=subprocess.DEVNULL, + stdout=log_handle, stderr=log_handle, + close_fds=True, + env=child_env, + **detached_process_kwargs(), + ) + finally: + log_handle.close() + return True + + +def recover_pending_handoffs() -> int: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + now = time.time() + for running in pending_dir.glob("*.running"): + try: + if now - running.stat().st_mtime > STALE_RUNNING_SECONDS: + running.replace(running.with_suffix(".json")) + except OSError: + continue + recoverable = [] + for handoff in pending_dir.glob("*.json"): + try: + age = now - handoff.stat().st_mtime + except OSError: + continue + if age > PENDING_EXPIRY_SECONDS: + handoff.unlink(missing_ok=True) + continue + recoverable.append((age, handoff)) + recoverable.sort(key=lambda item: item[0], reverse=True) + launched = 0 + for _, handoff in recoverable[:PENDING_LAUNCH_LIMIT]: + launched += int(_launch_handoff(handoff)) + return launched + + +def refresh_pending_handoffs() -> None: + pending_dir = data_dir() / "pending" + if not pending_dir.is_dir(): + return + for pattern in ("*.json", "*.running"): + for handoff in pending_dir.glob(pattern): + try: + os.utime(handoff) + except OSError: + continue + + +def hand_off_flush( + hook_input: dict, reason: str, *, wait_for_inflight: bool = False, +) -> None: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = ( + f"{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}\0{reason}" + ) + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + handoff_path = pending_dir / f"{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": reason, + "wait_for_inflight": wait_for_inflight, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + + +def automatic_flush_enabled() -> bool: + return os.environ.get("MEM0_CODE_AUTO_FLUSH", "true").lower() in { + "1", "true", "yes", "on", + } + + +def schedule_periodic_checkpoint( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + if ( + not automatic_flush_enabled() + or not api_key() + or not store.checkpoint_due(repo.identity, session_id) + ): + return False + if store.prepare_flush(repo, session_id, "periodic") is None: + return False + hand_off_flush(hook_input, "periodic") + return True + + +def _idle_flush_seconds() -> int: + try: + return max( + int(os.environ.get("MEM0_CODE_IDLE_FLUSH_SECONDS", str(DEFAULT_IDLE_FLUSH_SECONDS))), + 0, + ) + except ValueError: + return DEFAULT_IDLE_FLUSH_SECONDS + + +def schedule_idle_flush( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + delay = _idle_flush_seconds() + if delay <= 0 or not automatic_flush_enabled() or not api_key(): + return False + if store.has_inflight_flush(repo.identity, session_id): + return False + if not store.has_unflushed_events(repo.identity, session_id): + return False + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = f"idle\0{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}" + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + for old in pending_dir.glob(f"idle-{digest}*"): + old.unlink(missing_ok=True) + handoff_path = pending_dir / f"idle-{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": "idle", + "delay_seconds": delay, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + return True + + +def log_failure(exc: Exception) -> None: + try: + log_path = data_dir() / "plugin-errors.log" + with log_path.open("a", encoding="utf-8") as handle: + handle.write(f"{time.time():.3f} {type(exc).__name__}: {exc}\n") + except OSError: + pass + + +def run( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> int: + if record_stop_fn is None: + record_stop_fn = default_record_stop + if automatic_flush_reasons is None: + automatic_flush_reasons = {"session-end"} + + base_actions = ["session-start", "user-prompt", "post-tool", "stop", "flush"] + all_actions = base_actions + list((extra_actions or {}).keys()) + + parser = argparse.ArgumentParser() + parser.add_argument("action", choices=all_actions) + parser.add_argument("--reason", default="manual") + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + args = parser.parse_args() + + if args.harness: + configure_harness(args.harness) + telemetry.init(harness=args.harness) + + if args.plugin_data_dir: + os.environ[data_dir_env] = args.plugin_data_dir + + cache_plugin_api_key() + if args.action == "session-start": + clear_stale_api_key_cache() + + hook_input = read_hook_input() + store = EvidenceStore() + try: + if store.is_paused(): + if args.action == "session-start": + refresh_pending_handoffs() + telemetry.record("session_start", paused=True) + telemetry.spawn_flush() + return 0 + + if args.action == "session-start": + if telemetry.is_first_run(): + telemetry.record("install") + recovered = recover_pending_handoffs() + record_session_start(store, hook_input) + if recovered: + telemetry.record("handoff_recovered", count=recovered) + telemetry.spawn_flush() + elif args.action == "user-prompt": + output = first_prompt_memory_output(store, hook_input) + if output: + print(json.dumps(output)) + elif args.action == "post-tool": + record_tool(store, hook_input) + elif args.action == "stop": + repo, session_id = record_stop_fn(store, hook_input) + if not schedule_periodic_checkpoint(store, hook_input, repo, session_id): + schedule_idle_flush(store, hook_input, repo, session_id) + elif args.action == "flush": + automatic = args.reason in automatic_flush_reasons + if automatic and not automatic_flush_enabled(): + return 0 + if args.reason == "session-end": + record_stop_fn(store, hook_input) + if os.environ.get("MEM0_CODE_SYNC_FLUSH") == "1": + print(json.dumps(checkpoint_session(store, hook_input, args.reason))) + else: + session_id = str(hook_input.get("session_id") or "unknown-session") + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + already_running = store.has_inflight_flush(repo.identity, session_id) + if already_running and args.reason == "session-end": + hand_off_flush(hook_input, args.reason, wait_for_inflight=True) + elif not already_running and store.prepare_flush( + repo, session_id, args.reason, + ) is not None: + hand_off_flush(hook_input, args.reason) + elif extra_actions and args.action in extra_actions: + result = extra_actions[args.action](store, hook_input) + if result: + print(json.dumps(result)) + finally: + store.close() + return 0 + + +def entry_point( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> None: + try: + raise SystemExit(run( + record_stop_fn=record_stop_fn, + extra_actions=extra_actions, + data_dir_env=data_dir_env, + automatic_flush_reasons=automatic_flush_reasons, + )) + except Exception as exc: + log_failure(exc) + raise SystemExit(0) + + +if __name__ == "__main__": + entry_point() diff --git a/integrations/cursor-plugin/core/mcp_server.py b/integrations/cursor-plugin/core/mcp_server.py new file mode 100644 index 000000000..036fbbdc9 --- /dev/null +++ b/integrations/cursor-plugin/core/mcp_server.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Expose Mem0's memory search as one local coding-agent tool.""" + +from __future__ import annotations + +import json +import os +import sys +from typing import Any + +import telemetry +from memory_core import ( + CODING_MEMORY_CATEGORY_NAMES, + PLUGIN_VERSION, + SEARCH_SCOPES, + format_search_result, + resolve_repo, + search_memories, +) + +PROTOCOL_VERSION = "2024-11-05" +TOOL_NAME = "search_memories" +TOOL_DESCRIPTION = ( + "Search memories from earlier work in this repository. ALWAYS call this " + "tool before answering anything that could depend on prior context: the " + "user's preferences, facts about this codebase, history, people, projects, " + "or earlier decisions. Do not rely on the chat window alone. The " + "repository's memory is shared by everyone who works in it and includes " + "what it took to run, test, or build here, so search before assuming an " + "invocation works. The scope argument changes what is searched: 'repo' " + "(default) is the whole repository's shared memory plus your own " + "preferences, 'dir' narrows the shared part to the directory you are " + "working in, and 'mine' is your preferences alone." +) +TOOL_SCHEMA = { + "type": "object", + "properties": { + "query": { + "type": "string", + "minLength": 1, + "maxLength": 2000, + "description": "A direct question about earlier work in this repository.", + }, + "top_k": { + "type": "integer", + "minimum": 1, + "maximum": 20, + "description": "Maximum memories to return. Uses Mem0's configured default when omitted.", + }, + "category": { + "type": "string", + "enum": list(CODING_MEMORY_CATEGORY_NAMES), + "description": "Optional memory category. Omit to search every category.", + }, + "scope": { + "type": "string", + "enum": list(SEARCH_SCOPES), + "description": ( + "Which memories to search. 'repo' (default) is the whole repository's " + "shared memory plus your own preferences, 'dir' narrows the shared " + "part to the current directory, 'mine' is your preferences alone." + ), + }, + "run_id": { + "type": "string", + "minLength": 1, + "description": ( + "Optional coding-agent session ID. With any scope, restricts results to memories " + "saved in that session. Omit to recall memories across sessions." + ), + }, + }, + "required": ["query"], + "additionalProperties": False, +} + + +class ToolInputError(ValueError): + pass + + +def _validate_arguments( + arguments: Any, +) -> tuple[str, int | None, str | None, str | None, str | None]: + if not isinstance(arguments, dict): + raise ToolInputError("Search arguments must be an object.") + + unknown = set(arguments) - {"query", "top_k", "category", "scope", "run_id"} + if unknown: + raise ToolInputError(f"Unknown search argument: {sorted(unknown)[0]}") + + query = arguments.get("query") + if not isinstance(query, str) or not query.strip(): + raise ToolInputError("query must be a non-empty string.") + query = query.strip() + if len(query) > 2000: + raise ToolInputError("query must be at most 2,000 characters.") + + top_k = arguments.get("top_k") + if top_k is not None and ( + isinstance(top_k, bool) or not isinstance(top_k, int) or not 1 <= top_k <= 20 + ): + raise ToolInputError("top_k must be an integer from 1 to 20.") + + category = arguments.get("category") + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ToolInputError("category must be one of Mem0's supported categories.") + + scope = arguments.get("scope") + if scope is not None and scope not in SEARCH_SCOPES: + raise ToolInputError(f"scope must be one of {list(SEARCH_SCOPES)}.") + + run_id = arguments.get("run_id") + if run_id is not None: + if not isinstance(run_id, str) or not run_id.strip(): + raise ToolInputError("run_id must be a non-empty string.") + run_id = run_id.strip() + + return query, top_k, category, scope, run_id + + +def call_search_memories(arguments: Any, cwd: str | None = None) -> str: + query, top_k, category, scope, run_id = _validate_arguments(arguments) + repo = resolve_repo(cwd or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + result = search_memories( + None, + repo, + None, + query, + top_k=top_k, + category=category, + scope=scope, + run_id=run_id, + operation="mcp-search", + ) + return format_search_result(result) + + +def _workspace_cwd(params: dict[str, Any]) -> str | None: + meta = params.get("_meta") + if not isinstance(meta, dict): + return None + metadata = meta.get("x-codex-turn-metadata") + if not isinstance(metadata, dict): + return None + workspaces = metadata.get("workspaces") or {} + if isinstance(workspaces, dict): + return next((path for path in workspaces if isinstance(path, str) and path), None) + return None + + +def _tool_response(text: str, *, is_error: bool = False) -> dict[str, Any]: + return { + "content": [{"type": "text", "text": text}], + "isError": is_error, + } + + +def handle_request(message: Any) -> dict[str, Any] | None: + if not isinstance(message, dict): + return None + request_id = message.get("id") + method = message.get("method") + + if method == "notifications/initialized": + return None + if method == "initialize": + requested = (message.get("params") or {}).get("protocolVersion") + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "protocolVersion": requested or PROTOCOL_VERSION, + "capabilities": {"tools": {"listChanged": False}}, + "serverInfo": {"name": "mem0", "version": PLUGIN_VERSION}, + }, + } + if method == "ping": + return {"jsonrpc": "2.0", "id": request_id, "result": {}} + if method == "tools/list": + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "tools": [ + { + "name": TOOL_NAME, + "description": TOOL_DESCRIPTION, + "inputSchema": TOOL_SCHEMA, + "annotations": { + "readOnlyHint": True, + "idempotentHint": True, + "openWorldHint": True, + }, + } + ] + }, + } + if method == "tools/call": + params = message.get("params") or {} + if params.get("name") != TOOL_NAME: + result = _tool_response("Unknown Mem0 tool.", is_error=True) + else: + try: + result = _tool_response( + call_search_memories(params.get("arguments"), _workspace_cwd(params)) + ) + except ToolInputError as exc: + result = _tool_response(str(exc), is_error=True) + except Exception: + result = _tool_response("Memory search failed.", is_error=True) + return {"jsonrpc": "2.0", "id": request_id, "result": result} + if request_id is None: + return None + return { + "jsonrpc": "2.0", + "id": request_id, + "error": {"code": -32601, "message": "Method not found"}, + } + + +def main() -> int: + for raw_line in sys.stdin: + try: + message = json.loads(raw_line) + response = handle_request(message) + except json.JSONDecodeError: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32700, "message": "Parse error"}, + } + except Exception: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32603, "message": "Internal error"}, + } + if response is not None: + sys.stdout.write(json.dumps(response, separators=(",", ":")) + "\n") + sys.stdout.flush() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/cursor-plugin/core/memory_cli.py b/integrations/cursor-plugin/core/memory_cli.py new file mode 100644 index 000000000..595729feb --- /dev/null +++ b/integrations/cursor-plugin/core/memory_cli.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Mem0 diagnostics and user controls.""" + +from __future__ import annotations + +import argparse +import json +import os + +import telemetry +from memory_core import ( + EvidenceStore, + api_key, + data_dir, + doctor, + forget_remote_repo, + configure_harness, + resolve_repo, + user_id, +) + + +def _print_status(value: dict) -> None: + last = value.get("last_operation") or {} + print(f"Mem0: {'paused' if value['paused'] else 'active'}") + print(f"Repository: {value['repo_id']}") + print(f"Local data: {value['data_dir']}") + print(f"API key: {'configured' if value['api_key_configured'] else 'missing'}") + print( + "Saved on this computer: " + f"{value['events']} session details, {value['flushes']} memory updates" + ) + print( + f"Used in this repository: {value['retrievals']} memories returned, " + f"{value['sidekick_runs']} sidekick runs" + ) + if last: + item_label = "" + if last["operation"] in {"flush", "flush-retry"}: + item_label = f", {last['item_count']} memories" + operation = ( + "memory update" + if last["operation"] in {"flush", "flush-retry"} + else last["operation"].replace("-", " ") + ) + print( + f"Last {operation}: " + f"{'succeeded' if last['success'] else 'failed'} " + f"({last['duration_ms']:.1f} ms{item_label})" + ) + sidekick = value.get("last_sidekick") or {} + if sidekick: + state = "finished" if sidekick.get("stopped_at") else "started" + print( + "Last sidekick: " + f"{state}, received {sidekick['context_chars']} characters of memory, " + f"agent {sidekick['agent_id']}" + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + subparsers = parser.add_subparsers(dest="command", required=True) + + status = subparsers.add_parser("status") + status.add_argument("--json", action="store_true") + + doctor_parser = subparsers.add_parser("doctor") + doctor_parser.add_argument("--json", action="store_true") + + subparsers.add_parser("pause") + subparsers.add_parser("resume") + + forget = subparsers.add_parser("forget") + forget.add_argument("--remote", action="store_true") + forget.add_argument("--yes", action="store_true") + forget.add_argument("--include-project-memory", action="store_true") + + args = parser.parse_args() + if args.harness: + source_tag = f"{args.harness.replace('-', '_')}_plugin" + configure_harness(args.harness, source_tag=source_tag) + telemetry.init(harness=args.harness, source_tag=source_tag.upper()) + if args.plugin_data_dir: + os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir + store = EvidenceStore() + try: + repo = resolve_repo(os.getcwd()) + telemetry.record("control", repo=repo, action=args.command) + if args.command == "status": + result = { + **store.status(repo.identity), + "repo_id": repo.identity, + "app_id": repo.app_id, + "project_id": repo.project_id, + "directory": repo.directory, + "user_id": user_id(), + "data_dir": str(data_dir()), + "api_key_configured": bool(api_key()), + } + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + _print_status(result) + elif args.command == "doctor": + result = doctor(os.getcwd()) + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + for name, check in result["checks"].items(): + print( + f"{'PASS' if check['ok'] else 'FAIL'} {name}: {check['detail']}" + ) + return 0 if result["ok"] else 1 + elif args.command == "pause": + store.set_setting("paused", "true") + print("Mem0 stopped saving and searching memories.") + elif args.command == "resume": + store.set_setting("paused", "false") + print("Mem0 resumed saving and searching memories.") + elif args.command == "forget": + if not args.yes: + print( + "Refusing to delete data without --yes. Add --remote to also " + "delete this user/repository scope from Mem0." + ) + return 2 + remote_result = ( + forget_remote_repo( + repo, include_project_memory=args.include_project_memory + ) + if args.remote + else None + ) + local_result = store.forget_local_repo(repo.identity) + print( + json.dumps( + {"local": local_result, "remote": remote_result}, + indent=2, + default=str, + ) + ) + if remote_result and remote_result.get("status") == "error": + return 1 + finally: + store.close() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/cursor-plugin/core/memory_core.py b/integrations/cursor-plugin/core/memory_core.py new file mode 100644 index 000000000..cf71196b8 --- /dev/null +++ b/integrations/cursor-plugin/core/memory_core.py @@ -0,0 +1,2645 @@ +#!/usr/bin/env python3 +"""Shared core for Mem0 agent plugins. + +Hooks record small session details locally. When the agent compacts or ends the +session, Mem0 sends the useful parts to the platform so it can create memories. +The agent can search those memories during later work in the repository. +""" + +from __future__ import annotations + +import functools +import hashlib +import json +import math +import os +import re +import sqlite3 +import subprocess +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +import telemetry + +DEFAULT_API_URL = "https://api.mem0.ai" +PLUGIN_VERSION = "0.3.1" + +_harness_name: str = "generic" +_harness_env_prefix: str = "MEM0_PLUGIN" +_harness_data_dir_name: str = "mem0-plugin" +_harness_source_tag: str = "mem0_plugin" + + +def configure_harness( + name: str, + env_prefix: str = "", + data_dir_name: str = "", + source_tag: str = "", +) -> None: + global _harness_name, _harness_env_prefix, _harness_data_dir_name, _harness_source_tag + _harness_name = name + _harness_env_prefix = env_prefix or f"MEM0_{name.upper().replace('-', '_')}" + _harness_data_dir_name = data_dir_name or f"{name}-plugin" + _harness_source_tag = source_tag or f"{name.replace('-', '_')}_plugin" + + +def harness_config() -> dict[str, str]: + return { + "name": _harness_name, + "env_prefix": _harness_env_prefix, + "data_dir_name": _harness_data_dir_name, + "source_tag": _harness_source_tag, + } + + +MAX_COMMAND_CHARS = 2000 +MAX_RESULT_CHARS = 2500 +MAX_EPISODE_CHARS = 12000 +CHECKPOINT_EXCHANGES = 5 +CHECKPOINT_MESSAGES = 10 +CHECKPOINT_SOURCE_CHARS = 40000 +DEFAULT_MAX_CONTEXT_CHARS = 4000 +MAX_EXTRACTION_INPUT_TOKENS = 24000 +MAX_FLUSH_ATTEMPTS = 5 +FORGET_PAGE_SIZE = 100 +FORGET_MAX_PAGES = 50 + +PROJECT_MEMORY_INSTRUCTIONS = """Save concise repository facts that will help anyone with future coding work in this repository. + +A completed change should produce one memory explaining the resulting behavior, where it is implemented when useful, and any important constraints or reasoning. Exploration or accepted decisions may produce separate memories only when they are independently useful. + +A command that failed and was then made to work should produce one memory naming the failing invocation, the error it returned, and the invocation that succeeded. Do not save one-off errors caused by an edit still in progress, transient network failures, or anything a rerun would fix on its own. + +Use the current coding agent's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or the coding agent completed them. Treat subagent responses as supporting repository evidence, not as decisions. + +Write about the repository, not the user, assistant, session, or task. Do not save personal preferences. Do not save a memory that only states which repository, branch, or directory the session worked in. Do not include test results, documentation updates, release notes, or temporary state. + +If nothing useful was established, return no memories.""" + +PERSONAL_MEMORY_INSTRUCTIONS = """Save concise facts about the user that will help in any repository: preferred tools, package managers, languages, coding style, review and communication preferences, and anything the user explicitly asked to be remembered about themselves. + +Write in the third person about the user, not about the repository, the assistant, the session, or the task. Do not save repository facts, project decisions, commands, or what was built. + +Never save that the user has no preferences or that nothing was learned. If nothing was learned about the user, return no memories.""" + +CODING_MEMORY_CATEGORIES = [ + { + "project_knowledge": ( + "What the project is and how its code, APIs, data, files, and " + "components work." + ) + }, + { + "decisions_and_constraints": ( + "Why an approach was chosen, what must remain true, and rules future " + "work must follow." + ) + }, + { + "workflows": ( + "How to run, test, debug, deploy, configure, or otherwise work on the " + "project." + ) + }, + { + "problems_and_fixes": ( + "Bugs, failures, known pitfalls, their causes, and how to fix or avoid " + "them." + ) + }, + { + "results": ( + "Outcomes and measurements from tests, benchmarks, experiments, or " + "investigations." + ) + }, +] +CODING_MEMORY_CATEGORY_NAMES = tuple( + category_name + for category in CODING_MEMORY_CATEGORIES + for category_name in category +) + +TEST_COMMAND_RE = re.compile( + r"(?:^|\s)(?:pytest|py\.test|jest|vitest|go\s+test|cargo\s+test|" + r"npm\s+(?:run\s+)?test|pnpm\s+(?:run\s+)?test|yarn\s+test|" + r"mvn\s+test|gradle\s+test|make\s+test)(?:\s|$)", + re.IGNORECASE, +) +BUILD_COMMAND_RE = re.compile( + r"(?:^|\s)(?:npm|pnpm|yarn)\s+(?:run\s+)?build(?:\s|$)|" + r"(?:^|\s)(?:cargo|go|mvn|gradle|make)\s+build(?:\s|$)", + re.IGNORECASE, +) + +SECRET_PATTERNS = [ + re.compile(r"(?i)(authorization\s*[:=]\s*(?:bearer|token)\s+)[^\s\"']+"), + re.compile( + r"(?i)((?:api[_-]?key|secret[_-]?access[_-]?key|session[_-]?token)\s*[:=]\s*)[^\s\"']+" + ), + re.compile( + r"(?i)((?:access[_-]?token|refresh[_-]?token|password|credential)" + r"\s*[:=]\s*)[^\s&\"']+" + ), + re.compile(r"\b(?:sk|m0|mem0_sk|psk)-[A-Za-z0-9_\-]{12,}\b"), + re.compile(r"\b(?:ASIA|AKIA)[A-Z0-9]{12,}\b"), + re.compile(r"\b(?:ghp_|github_pat_|xox[baprs]-)[A-Za-z0-9_\-]{12,}\b"), + re.compile( + r"-----BEGIN [^-]*PRIVATE KEY-----.*?-----END [^-]*PRIVATE KEY-----", + re.DOTALL, + ), + re.compile( + r'(?i)("(?:api[_-]?key|password|secret(?:[_-]?access[_-]?key)?' + r'|(?:access|refresh|session)[_-]?token|token|authorization|credential' + r')"\s*:\s*")(?:\\.|[^"\\])*' + ), +] + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def redact(value: Any) -> str: + text = ( + value + if isinstance(value, str) + else json.dumps(value, ensure_ascii=False, default=str) + ) + for pattern in SECRET_PATTERNS: + if pattern.groups: + text = pattern.sub(r"\1[REDACTED]", text) + else: + text = pattern.sub("[REDACTED]", text) + return text + + +def bounded(value: Any, limit: int) -> str: + text = redact(value).strip() + if len(text) <= limit: + return text + return text[:limit] + f"\n...[truncated {len(text) - limit} chars]" + + +def _git(cwd: str, *args: str) -> str: + try: + result = subprocess.run( + ["git", "-C", cwd, *args], + check=False, + capture_output=True, + text=True, + timeout=0.5, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return result.stdout.strip() if result.returncode == 0 else "" + + +def _normalize_remote(remote: str) -> str: + remote = remote.strip() + if remote.startswith("git@") and ":" in remote: + host_path = remote[4:].replace(":", "/", 1) + remote = f"https://{host_path}" + if remote.endswith(".git"): + remote = remote[:-4] + if "://" in remote: + parsed = urllib.parse.urlsplit(remote) + hostname = parsed.hostname or "" + if parsed.port: + hostname = f"{hostname}:{parsed.port}" + remote = urllib.parse.urlunsplit( + (parsed.scheme, hostname, parsed.path, parsed.query, parsed.fragment) + ) + return remote.rstrip("/") + + +_WILDCARD_SCOPE = re.compile(r"^\*+$") + + +def _scope_value(raw: str | None) -> str: + """Reject wildcards as identities: they are filter syntax and would widen the scope.""" + value = (raw or "").strip() + return "" if _WILDCARD_SCOPE.match(value) else value + + +SEARCH_SCOPES = ("repo", "dir", "mine") +DEFAULT_SEARCH_SCOPE = "repo" + +def directory_app_id(repo: RepoContext) -> str: + """The app_id of the directory this session runs in: the repository at the root, repository/path below it.""" + return f"{repo.app_id}/{repo.directory}" if repo.directory else repo.app_id + + +def directory_chain(repo: RepoContext) -> list[str]: + """Every directory a memory belongs to, from the top-level folder down to the one it was written in.""" + parts = repo.directory.split("/") if repo.directory else [] + return ["/".join(parts[: index + 1]) for index in range(len(parts))] + + +def _shared_project_ids(repo: RepoContext) -> list[str]: + """Current and pre-upgrade namespaces, shared by recall and explicit deletion.""" + if not repo.identity.startswith("local:") and repo.project_id != repo.app_id: + return [repo.project_id, repo.app_id] + return [repo.project_id] + + +def _search_filters(user: str, repo: RepoContext, scope: str) -> dict[str, Any]: + """Build the scope filter: app_id scopes to the repo, then union shared and personal lanes.""" + app_scope = {"app_id": repo.app_id} + mine = {"AND": [{"user_id": user}, app_scope]} + if scope == "mine": + return mine + projects = [{"AND": [{"agent_id": project_id}, app_scope]} for project_id in _shared_project_ids(repo)] + shared: dict[str, Any] = projects[0] if len(projects) == 1 else {"OR": projects} + if scope == "dir" and repo.directory: + shared = {"AND": [shared, {"metadata": {"dirs": {"contains": repo.directory}}}]} + return {"OR": [shared, mine]} + + +def search_scope() -> str: + configured = ( + _plugin_option("search_scope", "MEM0_CODE_SEARCH_SCOPE") or "" + ).strip().lower() + return configured if configured in SEARCH_SCOPES else DEFAULT_SEARCH_SCOPE + + +def resolve_search_scope(scope: str | None) -> str: + value = (scope or search_scope()).strip().lower() + if value not in SEARCH_SCOPES: + raise ValueError(f"Unknown search scope: {value}") + return value + + +def _legacy_project_map(cwd: str, root: str, raw_remote: str) -> str: + """Return the project name used by the previous Claude Code plugin.""" + try: + data = json.loads((Path.home() / ".mem0" / "project_map.json").read_text()) + except (OSError, json.JSONDecodeError): + return "" + if not isinstance(data, dict): + return "" + + keys = list(dict.fromkeys([cwd, root, os.path.realpath(cwd), os.path.realpath(root)])) + if raw_remote: + keys.append(f"remote:{hashlib.sha256(raw_remote.encode()).hexdigest()[:16]}") + for key in keys: + value = data.get(key) + if isinstance(value, str) and value.strip(): + return value.strip() + return "" + + +def _legacy_project_id(cwd: str, root: str, raw_remote: str, identity: str) -> str: + """Use the repository namespace created by the previous Mem0 plugin.""" + configured = _scope_value(os.environ.get("MEM0_PROJECT_ID")) + if configured: + return configured + + mapped = _scope_value(_legacy_project_map(cwd, root, raw_remote)) + if mapped: + return mapped + + remote = raw_remote or ("" if identity.startswith("local:") else identity) + remote = remote.strip().removesuffix(".git") + for prefix in ("https://", "http://", "ssh://", "git://"): + if remote.startswith(prefix): + remote = remote[len(prefix) :] + break + else: + remote = re.sub(r"^git@", "", remote) + parts = [part for part in remote.replace(":", "/", 1).split("/") if part] + if len(parts) >= 2: + return f"{parts[-2]}-{parts[-1]}".replace("/", "-").replace(":", "-") + if parts: + return parts[-1].replace("/", "-").replace(":", "-") + return os.path.basename(root or cwd) or "unknown" + + +@dataclass(frozen=True) +class RepoContext: + cwd: str + root: str + identity: str + app_id: str + branch: str + head_sha: str + project_id: str = "" + directory: str = "" + + +def _project_id(root: str, identity: str, app_id: str) -> str: + """The shared namespace: includes a host hash so repos with the same owner/name on different hosts stay apart.""" + if not identity.startswith("local:"): + return f"{app_id}-{hashlib.sha256(identity.encode()).hexdigest()[:10]}" + return f"local-{app_id}-{hashlib.sha256(root.encode()).hexdigest()[:10]}" + + +def _relative_directory(cwd: str, root: str) -> str: + relative = os.path.relpath(cwd, root) + return "" if relative == "." or relative.startswith("..") else relative.replace(os.sep, "/") + + +@dataclass(frozen=True) +class MemorySearchResult: + succeeded: bool + matched_count: int + already_shown_count: int + memories: list[dict[str, Any]] + + +@functools.lru_cache(maxsize=64) +def _resolve_repo_cached(cwd: str) -> RepoContext: + given_cwd = cwd + cwd = os.path.realpath(cwd) + given_root = _git(cwd, "rev-parse", "--show-toplevel") or given_cwd + root = os.path.realpath(given_root) + raw_remote = _git(root, "config", "--get", "remote.origin.url") + remote = _normalize_remote(raw_remote) + identity = remote or f"local:{root}" + app_id = _legacy_project_id(given_cwd, given_root, raw_remote, identity) + return RepoContext( + cwd=cwd, + root=root, + identity=identity, + app_id=app_id, + branch=_git(root, "branch", "--show-current") or "detached", + head_sha=_git(root, "rev-parse", "HEAD"), + project_id=_project_id(root, identity, app_id), + directory=_relative_directory(cwd, root), + ) + + +def resolve_repo(cwd: str | None) -> RepoContext: + return _resolve_repo_cached(os.path.abspath(cwd or os.getcwd())) + + +def api_key() -> str: + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return configured + try: + return (data_dir() / "api-key").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def cache_plugin_api_key() -> bool: + """Bridge host's hook-only sensitive config into plugin-owned storage.""" + configured = ( + os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if not configured: + return False + + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + path = directory / "api-key" + temporary = directory / f"api-key.{os.getpid()}.tmp" + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_TRUNC, + 0o600, + ) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + handle.write(configured) + os.replace(temporary, path) + os.chmod(path, 0o600) + finally: + try: + temporary.unlink() + except FileNotFoundError: + pass + return True + + +def clear_stale_api_key_cache() -> bool: + """Drop the cached key file once every configured key source is gone.""" + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return False + path = data_dir() / "api-key" + if not path.exists(): + return False + try: + path.unlink() + except OSError: + return False + return True + + +def detached_process_kwargs(platform: str | None = None) -> dict: + """Keep a spawned worker alive after the coding agent exits, on POSIX and Windows.""" + if (platform or sys.platform) == "win32": + return { + "creationflags": subprocess.DETACHED_PROCESS + | subprocess.CREATE_NEW_PROCESS_GROUP + } + return {"start_new_session": True} + + +def _plugin_option(name: str, fallback: str = "") -> str: + return ( + os.environ.get(f"PLUGIN_OPTION_{name.upper()}") + or os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") + or os.environ.get(fallback) + or "" + ).strip() + + +def user_id() -> str: + return ( + _scope_value(_plugin_option("user_id", "MEM0_CODE_USER_ID")) + or _scope_value(os.environ.get("MEM0_USER_ID")) + or _scope_value(os.environ.get("MEM0_RESOLVED_USER_ID")) + or _scope_value(os.environ.get("USER")) + or _scope_value(os.environ.get("USERNAME")) + or "default" + ) + + +def data_dir() -> Path: + configured = ( + os.environ.get("MEM0_CODE_DATA_DIR") + or os.environ.get("MEM0_PLUGIN_DATA_DIR") + or os.environ.get("PLUGIN_DATA") + or os.environ.get("CLAUDE_PLUGIN_DATA") + ) + return ( + Path(configured).expanduser() if configured else Path.home() / ".mem0" / _harness_data_dir_name + ) + + +def _bool_option(name: str, fallback: str, default: bool = False) -> bool: + value = _plugin_option(name, fallback) + if not value: + return default + return value.lower() in {"1", "true", "yes", "on"} + + +def _int_option(name: str, fallback: str, default: int) -> int: + value = _plugin_option(name, fallback) + try: + return int(value) if value else default + except ValueError: + return default + + + +def _checkpoint_message(event: dict[str, Any]) -> str: + kind = event.get("kind") + payload = event.get("payload") or {} + if kind == "user_prompt": + return redact(payload.get("text", "")).strip() + if kind == "assistant_stop": + transcript_messages = payload.get("transcript_messages") or [] + if isinstance(transcript_messages, list): + text = "\n".join( + str(message.get("content") or "") + for message in transcript_messages + if isinstance(message, dict) and message.get("content") + ) + if text: + return text + return redact(payload.get("text", "")).strip() + if kind == "sidekick_stop": + return redact(payload.get("final_message", "")).strip() + return "" + + +def checkpoint_stats(events: list[dict[str, Any]]) -> tuple[int, int, int]: + """Return completed exchanges, messages, and source characters.""" + completed = sum(event.get("kind") == "assistant_stop" for event in events) + contents = [content for event in events if (content := _checkpoint_message(event))] + return completed, len(contents), sum(len(content) for content in contents) + + +def select_checkpoint_events( + events: list[dict[str, Any]], *, force: bool +) -> list[dict[str, Any]]: + """Select one ordered extraction block without splitting an exchange.""" + for index, event in enumerate(events): + if event.get("kind") != "assistant_stop": + continue + candidate = events[: index + 1] + completed, messages, source_chars = checkpoint_stats(candidate) + if ( + completed >= CHECKPOINT_EXCHANGES + or messages >= CHECKPOINT_MESSAGES + or source_chars >= CHECKPOINT_SOURCE_CHARS + ): + return candidate + return events if force else [] + + +class EvidenceStore: + def __init__(self, path: Path | None = None): + directory = data_dir() if path is None else path.parent + directory.mkdir(parents=True, exist_ok=True) + self.path = path or directory / "evidence.sqlite3" + try: + self._open() + except sqlite3.DatabaseError: + self._quarantine() + self._open() + + def _open(self) -> None: + self.conn = sqlite3.connect(self.path, timeout=10) + self.conn.row_factory = sqlite3.Row + try: + self.conn.execute("PRAGMA journal_mode=WAL") + self.conn.execute("PRAGMA busy_timeout=10000") + self._migrate() + except sqlite3.DatabaseError: + self.conn.close() + raise + + def _quarantine(self) -> None: + """Move an unreadable database aside so capture restarts cleanly.""" + stamp = int(time.time()) + for suffix in ("", "-wal", "-shm"): + source = Path(f"{self.path}{suffix}") + try: + source.replace(f"{self.path}.corrupt-{stamp}{suffix}") + except FileNotFoundError: + continue + except OSError: + try: + source.unlink() + except OSError: + pass + telemetry.record("db_quarantined") + + def close(self) -> None: + self.conn.close() + + def _migrate(self) -> None: + self.conn.executescript( + """ + CREATE TABLE IF NOT EXISTS events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + created_at TEXT NOT NULL, + kind TEXT NOT NULL, + payload_json TEXT NOT NULL, + flush_id TEXT + ); + CREATE INDEX IF NOT EXISTS events_session_idx + ON events(repo_id, session_id, flush_id, id); + + CREATE TABLE IF NOT EXISTS session_scopes ( + session_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + root TEXT NOT NULL, + branch TEXT NOT NULL, + head_sha TEXT NOT NULL, + created_at TEXT NOT NULL, + directory TEXT NOT NULL DEFAULT '' + ); + + CREATE TABLE IF NOT EXISTS flushes ( + packet_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + reason TEXT NOT NULL, + event_start INTEGER NOT NULL, + event_end INTEGER NOT NULL, + status TEXT NOT NULL, + episode_event_id TEXT, + semantic_event_id TEXT, + error TEXT, + attempts INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + CREATE TABLE IF NOT EXISTS retrievals ( + session_id TEXT NOT NULL, + repo_id TEXT NOT NULL, + memory_id TEXT NOT NULL, + injected_at TEXT NOT NULL, + rank INTEGER, + score REAL, + memory_text TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(session_id, repo_id, memory_id) + ); + + CREATE TABLE IF NOT EXISTS operations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + created_at TEXT NOT NULL, + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + operation TEXT NOT NULL, + duration_ms REAL NOT NULL, + success INTEGER NOT NULL, + item_count INTEGER NOT NULL DEFAULT 0, + request_chars INTEGER NOT NULL DEFAULT 0, + response_chars INTEGER NOT NULL DEFAULT 0, + error TEXT + ); + + CREATE TABLE IF NOT EXISTS sidekick_runs ( + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + agent_type TEXT NOT NULL, + started_at TEXT NOT NULL, + stopped_at TEXT, + transcript_path TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + final_message TEXT, + PRIMARY KEY(repo_id, session_id, agent_id) + ); + CREATE INDEX IF NOT EXISTS sidekick_runs_repo_idx + ON sidekick_runs(repo_id, started_at); + + CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + """ + ) + # Remove the pre-0.1.1 no-tools snapshot implementation. The real coding + # sidekick is a native Claude Code agent and stores no state in this DB. + self.conn.executescript( + """ + DROP TABLE IF EXISTS sidekick_calls; + DROP TABLE IF EXISTS sidekick_snapshots; + DROP TABLE IF EXISTS sidekick_state; + DROP TABLE IF EXISTS sidekick_packets; + """ + ) + self._ensure_column("retrievals", "rank", "INTEGER") + self._ensure_column("retrievals", "score", "REAL") + self._ensure_column("retrievals", "memory_text", "TEXT") + self._ensure_column("retrievals", "context_chars", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("flushes", "attempts", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("session_scopes", "directory", "TEXT NOT NULL DEFAULT ''") + self.conn.commit() + + def _ensure_column(self, table: str, column: str, declaration: str) -> None: + columns = { + str(row["name"]) + for row in self.conn.execute(f"PRAGMA table_info({table})").fetchall() + } + if column not in columns: + self.conn.execute(f"ALTER TABLE {table} ADD COLUMN {column} {declaration}") + + def record_event( + self, + repo: RepoContext, + session_id: str, + kind: str, + payload: dict[str, Any], + ) -> int: + cursor = self.conn.execute( + """INSERT INTO events + (repo_id, app_id, session_id, created_at, kind, payload_json) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + repo.app_id, + session_id, + utc_now(), + kind, + json.dumps(payload, ensure_ascii=False, sort_keys=True), + ), + ) + self.conn.commit() + return int(cursor.lastrowid) + + def record_assistant_response(self, repo: RepoContext, session_id: str, message: str) -> None: + """Ignore repeated response hooks until another prompt or a different answer arrives.""" + with self.conn: + # Serialize the check and insert across concurrent Stop and SessionEnd hooks. + self.conn.execute("BEGIN IMMEDIATE") + previous = self.conn.execute( + """SELECT kind, payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind IN ('user_prompt', 'assistant_stop') + ORDER BY id DESC LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if ( + previous is not None + and previous["kind"] == "assistant_stop" + and json.loads(previous["payload_json"]).get("text") == message + ): + return + self.record_event(repo, session_id, "assistant_stop", {"text": message}) + + def repo_for_session(self, session_id: str, cwd: str | None) -> RepoContext: + """Keep one project scope for every hook in a coding-agent session.""" + current = resolve_repo(cwd) + if session_id == "unknown-session": + return current + + with self.conn: + self.conn.execute( + """INSERT OR IGNORE INTO session_scopes + (session_id, repo_id, app_id, root, branch, head_sha, created_at, directory) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + current.identity, + current.app_id, + current.root, + current.branch, + current.head_sha, + utc_now(), + current.directory, + ), + ) + scope = self.conn.execute( + "SELECT * FROM session_scopes WHERE session_id = ?", (session_id,) + ).fetchone() + same_git_repo = ( + current.identity == scope["repo_id"] and bool(current.head_sha) + ) + pinned = current if same_git_repo else resolve_repo(str(scope["root"])) + return RepoContext( + cwd=current.cwd, + root=pinned.root, + identity=str(scope["repo_id"]), + app_id=str(scope["app_id"]), + branch=pinned.branch, + head_sha=pinned.head_sha, + project_id=pinned.project_id, + directory=str(scope["directory"] or ""), + ) + + def prepare_flush( + self, repo: RepoContext, session_id: str, reason: str + ) -> tuple[str, list[dict[str, Any]]] | None: + with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + existing = self.conn.execute( + """SELECT * FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') + ORDER BY created_at LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: + self.conn.execute( + "UPDATE flushes SET status = 'gave-up', updated_at = ? WHERE packet_id = ?", + (utc_now(), existing["packet_id"]), + ) + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status="gave-up", + success=False, + attempts=int(existing["attempts"] or 0), + ) + existing = None + if existing: + if reason != "periodic" and existing["reason"] == "periodic": + self.conn.execute( + "UPDATE flushes SET reason = ?, updated_at = ? WHERE packet_id = ?", + (reason, utc_now(), existing["packet_id"]), + ) + existing_rows = self.conn.execute( + "SELECT * FROM events WHERE flush_id = ? ORDER BY id", + (existing["packet_id"],), + ).fetchall() + if existing_rows: + return str(existing["packet_id"]), [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in existing_rows + ] + + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo.identity, session_id), + ).fetchall() + if not rows: + return None + + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + events = select_checkpoint_events(events, force=reason != "periodic") + if not events: + return None + event_start, event_end = events[0]["id"], events[-1]["id"] + packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" + packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] + now = utc_now() + + self.conn.execute( + """INSERT OR IGNORE INTO flushes + (packet_id, repo_id, app_id, session_id, reason, event_start, + event_end, status, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, 'prepared', ?, ?)""", + ( + packet_id, + repo.identity, + repo.app_id, + session_id, + reason, + event_start, + event_end, + now, + now, + ), + ) + event_ids = [event["id"] for event in events] + placeholders = ", ".join("?" for _ in event_ids) + self.conn.execute( + f"UPDATE events SET flush_id = ? " + f"WHERE id IN ({placeholders}) AND flush_id IS NULL", + (packet_id, *event_ids), + ) + return packet_id, events + + def checkpoint_due(self, repo_id: str, session_id: str) -> bool: + if self.has_inflight_flush(repo_id, session_id): + return False + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo_id, session_id), + ).fetchall() + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + return bool(select_checkpoint_events(events, force=False)) + + def has_inflight_flush(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status IN ('prepared', 'semantic-queued') + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def flush_record(self, packet_id: str) -> dict[str, Any] | None: + row = self.conn.execute( + "SELECT * FROM flushes WHERE packet_id = ?", (packet_id,) + ).fetchone() + return dict(row) if row else None + + def has_unflushed_events(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def unflushed_starts_with_session_start( + self, repo_id: str, session_id: str + ) -> bool: + row = self.conn.execute( + """SELECT kind FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id LIMIT 1""", + (repo_id, session_id), + ).fetchone() + return bool(row and row["kind"] == "session_start") + + def update_flush(self, packet_id: str, **fields: Any) -> None: + allowed = {"status", "episode_event_id", "semantic_event_id", "error"} + updates = {key: value for key, value in fields.items() if key in allowed} + updates["updated_at"] = utc_now() + clause = ", ".join(f"{key} = ?" for key in updates) + failed = str(fields.get("status", "")) in { + "error", + "semantic-failed", + "semantic-timeout", + "semantic-missing", + } + if failed: + clause += ", attempts = attempts + 1" + with self.conn: + self.conn.execute( + f"UPDATE flushes SET {clause} WHERE packet_id = ?", + [*updates.values(), packet_id], + ) + + def unseen( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> list[dict[str, Any]]: + seen = { + row["memory_id"] + for row in self.conn.execute( + "SELECT memory_id FROM retrievals WHERE session_id = ? AND repo_id = ?", + (session_id, repo_id), + ) + } + return [memory for memory in memories if str(memory.get("id", "")) not in seen] + + def mark_injected( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> None: + now = utc_now() + with self.conn: + for rank, memory in enumerate(memories, start=1): + memory_id = str(memory.get("id", "")) + if memory_id: + memory_text = bounded( + memory.get("memory") or memory.get("text") or "", + 4000, + ) + try: + score = float(memory["score"]) + except (KeyError, TypeError, ValueError): + score = None + self.conn.execute( + """INSERT OR IGNORE INTO retrievals + (session_id, repo_id, memory_id, injected_at, rank, + score, memory_text, context_chars) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + repo_id, + memory_id, + now, + rank, + score, + memory_text, + len(memory_text), + ), + ) + + def injected_memories( + self, session_id: str, repo_id: str + ) -> list[dict[str, Any]]: + """Return the exact memories already supplied to the main conversation.""" + rows = self.conn.execute( + """SELECT memory_id, rank, score, memory_text + FROM retrievals + WHERE session_id = ? AND repo_id = ? + ORDER BY COALESCE(rank, 2147483647), injected_at, memory_id""", + (session_id, repo_id), + ).fetchall() + return [ + { + "id": row["memory_id"], + "memory": row["memory_text"], + "score": row["score"], + } + for row in rows + if row["memory_text"] + ] + + def start_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + context_chars: int, + ) -> bool: + """Record one native sidekick instance and whether context was first sent.""" + with self.conn: + cursor = self.conn.execute( + """INSERT OR IGNORE INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + context_chars) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + utc_now(), + context_chars, + ), + ) + return int(cursor.rowcount) > 0 + + def stop_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + transcript_path: str, + final_message: str, + ) -> str: + now = utc_now() + self.conn.execute("BEGIN IMMEDIATE") + try: + if not agent_id: + rows = self.conn.execute( + """SELECT agent_id FROM sidekick_runs + WHERE repo_id = ? AND session_id = ? AND agent_type = ? AND stopped_at IS NULL + LIMIT 2""", + (repo.identity, session_id, agent_type), + ).fetchall() + # Without a host ID, overlapping runs cannot be correlated reliably. + agent_id = rows[0]["agent_id"] if len(rows) == 1 else f"unknown-agent-{time.time_ns()}" + self.conn.execute( + """INSERT INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + stopped_at, transcript_path, final_message) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(repo_id, session_id, agent_id) DO UPDATE SET + stopped_at = excluded.stopped_at, + transcript_path = excluded.transcript_path, + final_message = excluded.final_message""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + now, + now, + bounded(transcript_path, 2000), + redact(final_message).strip(), + ), + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + return agent_id + + def operation( + self, + repo: RepoContext, + session_id: str, + operation: str, + duration_ms: float, + success: bool, + *, + item_count: int = 0, + request_chars: int = 0, + response_chars: int = 0, + error: str = "", + ) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO operations + (created_at, repo_id, session_id, operation, duration_ms, + success, item_count, request_chars, response_chars, error) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + ( + utc_now(), + repo.identity, + session_id, + operation, + duration_ms, + int(success), + item_count, + request_chars, + response_chars, + bounded(error, 1000), + ), + ) + + def has_operation(self, repo_id: str, session_id: str, operation: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM operations + WHERE repo_id = ? AND session_id = ? AND operation = ? + LIMIT 1""", + (repo_id, session_id, operation), + ).fetchone() + return row is not None + + def has_event(self, repo_id: str, session_id: str, kind: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + return row is not None + + def latest_event_payload( + self, repo_id: str, session_id: str, kind: str + ) -> dict[str, Any]: + row = self.conn.execute( + """SELECT payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + ORDER BY id DESC LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + if not row: + return {} + try: + payload = json.loads(row["payload_json"]) + except json.JSONDecodeError: + return {} + return payload if isinstance(payload, dict) else {} + + def setting(self, key: str, default: str = "") -> str: + row = self.conn.execute( + "SELECT value FROM settings WHERE key = ?", (key,) + ).fetchone() + return str(row["value"]) if row else default + + def set_setting(self, key: str, value: str) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO settings(key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at""", + (key, value, utc_now()), + ) + + def is_paused(self) -> bool: + return self.setting("paused", "false").lower() in { + "1", + "true", + "yes", + "on", + } + + def forget_local_repo(self, repo_id: str) -> dict[str, int]: + tables = { + "events": "repo_id", + "session_scopes": "repo_id", + "flushes": "repo_id", + "retrievals": "repo_id", + "operations": "repo_id", + "sidekick_runs": "repo_id", + } + removed: dict[str, int] = {} + with self.conn: + for table, column in tables.items(): + cursor = self.conn.execute( + f"DELETE FROM {table} WHERE {column} = ?", (repo_id,) + ) + removed[table] = max(int(cursor.rowcount), 0) + return removed + + def status(self, repo_id: str) -> dict[str, Any]: + def count(table: str) -> int: + return int( + self.conn.execute( + f"SELECT COUNT(*) FROM {table} WHERE repo_id = ?", (repo_id,) + ).fetchone()[0] + ) + + last_operation = self.conn.execute( + """SELECT created_at, operation, duration_ms, success, item_count, error + FROM operations WHERE repo_id = ? ORDER BY id DESC LIMIT 1""", + (repo_id,), + ).fetchone() + last_sidekick = self.conn.execute( + """SELECT session_id, agent_id, agent_type, started_at, stopped_at, + context_chars + FROM sidekick_runs WHERE repo_id = ? + ORDER BY started_at DESC LIMIT 1""", + (repo_id,), + ).fetchone() + return { + "paused": self.is_paused(), + "events": count("events"), + "flushes": count("flushes"), + "retrievals": count("retrievals"), + "sidekick_runs": count("sidekick_runs"), + "last_operation": dict(last_operation) if last_operation else None, + "last_sidekick": dict(last_sidekick) if last_sidekick else None, + } + + +def _session_id(hook_input: dict[str, Any]) -> str: + return str(hook_input.get("session_id") or "unknown-session") + + +def record_session_start(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + store.record_event( + repo, + session_id, + "session_start", + { + "source": hook_input.get("source", "startup"), + "model": bounded(hook_input.get("model", ""), 200), + "branch": repo.branch, + "head_sha": repo.head_sha, + }, + ) + telemetry.record( + "session_start", + repo=repo, + session_id=session_id, + trigger=bounded(str(hook_input.get("source", "startup")), 60), + model=bounded(hook_input.get("model", ""), 200), + api_key_configured=bool(api_key()), + is_git_repo=not repo.identity.startswith("local:"), + ) + + +def record_user_prompt( + store: EvidenceStore, hook_input: dict[str, Any] +) -> tuple[RepoContext, str, str, bool]: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prompt = redact(hook_input.get("prompt", "")).strip() + is_first_prompt = not store.has_event(repo.identity, session_id, "user_prompt") + store.record_event(repo, session_id, "user_prompt", {"text": prompt}) + return repo, session_id, prompt, is_first_prompt + + +def _tool_result_preview(response: Any) -> str: + if isinstance(response, dict): + selected = {} + for key in ( + "stdout", + "stderr", + "output", + "content", + "error", + "filePath", + "success", + "interrupted", + ): + if key in response: + selected[key] = response[key] + response = selected or {"keys": sorted(response.keys())[:20]} + return bounded(response, MAX_RESULT_CHARS) + + +def tool_payload(hook_input: dict[str, Any], *, failed: bool | None = False) -> dict[str, Any]: + name = str(hook_input.get("tool_name") or "unknown") + tool_input = hook_input.get("tool_input") or {} + if not isinstance(tool_input, dict): + tool_input = {} + payload: dict[str, Any] = { + "tool": name, + "failed": failed, + "duration_ms": hook_input.get("duration_ms"), + "agent_role": "sidekick" if hook_input.get("agent_id") else "main", + } + if hook_input.get("agent_id"): + payload["agent_id"] = bounded(hook_input["agent_id"], 200) + if hook_input.get("agent_type"): + payload["agent_type"] = bounded(hook_input["agent_type"], 200) + + if name in {"Read", "Write", "Edit", "MultiEdit", "NotebookEdit"}: + path = tool_input.get("file_path") or tool_input.get("notebook_path") + if path: + payload["path"] = bounded(path, 1000) + if name in {"Write", "Edit", "MultiEdit", "NotebookEdit"}: + payload["mutation_chars"] = sum( + len(str(tool_input.get(key, ""))) + for key in ("content", "new_string", "new_source", "edits") + ) + elif name == "Bash" or "command" in tool_input: + command = bounded(tool_input.get("command", ""), MAX_COMMAND_CHARS) + payload["command"] = command + payload["command_kind"] = ( + "test" + if TEST_COMMAND_RE.search(command) + else "build" + if BUILD_COMMAND_RE.search(command) + else "shell" + ) + response = ( + hook_input.get("error") if failed else hook_input.get("tool_response") + ) + payload["result_preview"] = _tool_result_preview(response) + elif name in {"Grep", "Glob", "WebSearch", "WebFetch"}: + for key in ("pattern", "path", "query", "url"): + if tool_input.get(key): + payload[key] = bounded(tool_input[key], 1000) + else: + payload["input_keys"] = sorted(tool_input.keys())[:20] + if failed: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + + if failed and "error" not in payload: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + return payload + + +def record_tool( + store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool | None = False +) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + payload = tool_payload(hook_input, failed=failed) + if payload.get("path"): + payload["repo_path"] = _repo_relative_path(repo, str(payload["path"])) + store.record_event( + repo, + session_id, + "tool_failure" if failed else "tool_result", + payload, + ) + + +def record_sidekick_start( + store: EvidenceStore, hook_input: dict[str, Any], *, inject_context: bool = True +) -> str: + """Record a native sidekick and reuse the main turn's retrieved memories.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_id = bounded(hook_input.get("agent_id", "unknown-agent"), 200) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + context = combine_context( + format_context(store.injected_memories(session_id, repo.identity)) + ) + if not inject_context: + context = "" + first_start = store.start_sidekick( + repo, session_id, agent_id, agent_type, len(context) + ) + store.record_event( + repo, + session_id, + "sidekick_start", + { + "agent_id": agent_id, + "agent_type": agent_type, + "context_chars": len(context) if first_start else 0, + "worktree_root": bounded(repo.root, 2000), + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="start", + first_start=first_start, + context_chars=len(context) if first_start else 0, + ) + return context if first_start else "" + + +def record_sidekick_stop(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + agent_id = bounded(hook_input.get("agent_id", ""), 200) + final_message = redact(hook_input.get("last_assistant_message", "")).strip() + transcript_path = bounded(hook_input.get("agent_transcript_path", ""), 2000) + agent_id = store.stop_sidekick( + repo, + session_id, + agent_id, + agent_type, + transcript_path, + final_message, + ) + store.record_event( + repo, + session_id, + "sidekick_stop", + { + "agent_id": agent_id, + "agent_type": agent_type, + "transcript_path": transcript_path, + "final_message": final_message, + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="stop", + has_transcript=bool(transcript_path), + message_chars=len(final_message), + ) + + +def _ordered_unique(values: Iterable[str]) -> list[str]: + seen: set[str] = set() + result = [] + for value in values: + if value and value not in seen: + seen.add(value) + result.append(value) + return result + + +def _repo_relative_path(repo: RepoContext, value: str) -> str: + value = str(value or "").strip() + if not value: + return "" + try: + path = Path(value) + if path.is_absolute(): + try: + return path.resolve().relative_to(Path(repo.root).resolve()).as_posix() + except ValueError: + return "" + except (OSError, ValueError): + pass + return bounded(value, 1000) + + +def _render_command_lines(commands: list[dict[str, str]]) -> list[str]: + lines = [] + for command in commands: + line = f"- [{command['status']}/{command['kind']}] {command['command']}" + if command["result"]: + line += f" — {bounded(command['result'], 500).replace(chr(10), ' ')}" + lines.append(line) + return lines + + +def build_episode( + repo: RepoContext, + session_id: str, + packet_id: str, + events: list[dict[str, Any]], + *, + canonical_task: str = "", + task_outcome: str = "", +) -> tuple[str, dict[str, Any]]: + prompts = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "user_prompt" and e["payload"].get("text") + ] + assistant_conclusions = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "assistant_stop" and e["payload"].get("text") + ] + sidekick_outcomes = [ + redact(e["payload"].get("final_message", "")).strip() + for e in events + if e["kind"] == "sidekick_stop" and e["payload"].get("final_message") + ] + tools = [ + e["payload"] for e in events if e["kind"] in {"tool_result", "tool_failure"} + and e["payload"].get("agent_role", "main") == "main" + ] + read_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") == "Read" + ) + modified_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") in {"Write", "Edit", "MultiEdit", "NotebookEdit"} + ) + searches = [ + {key: t[key] for key in ("tool", "pattern", "path", "query", "url") if key in t} + for t in tools + if t.get("tool") in {"Grep", "Glob", "WebSearch", "WebFetch"} + ] + commands = [ + { + "command": t.get("command", ""), + "kind": t.get("command_kind", "shell"), + "status": "unknown" if t.get("failed", False) is None else "failed" if t.get("failed") else "succeeded", + "result": t.get("result_preview", ""), + } + for t in tools + if t.get("command") + ] + + task = bounded(canonical_task or (prompts[0] if prompts else ""), 4000) + conclusion = redact(assistant_conclusions[-1] if assistant_conclusions else "").strip() + outcome = bounded(task_outcome, 2000) + + extraction_messages: list[dict[str, str]] = [] + pending_user_messages: list[dict[str, str]] = [] + if task and not prompts: + pending_user_messages.append({"role": "user", "content": task}) + for event in events: + if event["kind"] == "user_prompt" and event["payload"].get("text"): + pending_user_messages.append( + { + "role": "user", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + elif event["kind"] == "assistant_stop": + transcript_messages = event["payload"].get("transcript_messages") or [] + if isinstance(transcript_messages, list) and transcript_messages: + transcript_users = { + redact(message.get("content") or "").strip() + for message in transcript_messages + if isinstance(message, dict) and message.get("role") == "user" + } + extraction_messages.extend( + message + for message in pending_user_messages + if message["content"].strip() not in transcript_users + ) + extraction_messages.extend( + { + "role": str(message.get("role") or ""), + "content": redact(message.get("content") or "").strip(), + } + for message in transcript_messages + if isinstance(message, dict) + and message.get("role") in {"user", "assistant"} + and message.get("content") + ) + else: + extraction_messages.extend(pending_user_messages) + if event["payload"].get("text"): + extraction_messages.append( + { + "role": "assistant", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + pending_user_messages = [] + elif event["kind"] == "sidekick_stop": + pass + extraction_messages.extend(pending_user_messages) + + structured = { + "packet_id": packet_id, + "repo": repo.identity, + "app_id": repo.app_id, + "session_id": session_id, + "branch": repo.branch, + "head_sha": repo.head_sha, + "task": task, + "task_outcome": outcome, + "assistant_conclusion": conclusion, + "user_messages": prompts, + "assistant_outcomes": assistant_conclusions, + "sidekick_outcomes": sidekick_outcomes, + "extraction_messages": extraction_messages, + "files_read": read_paths[:50], + "files_modified": modified_paths[:50], + "searches": searches[-30:], + "commands": commands[-30:], + } + lines = ["Coding-session episode"] + if task: + lines.extend(["", "Task:", task]) + if modified_paths: + lines.extend( + ["", "Files modified:", *[f"- {path}" for path in modified_paths[:50]]] + ) + if read_paths: + lines.extend(["", "Files read:", *[f"- {path}" for path in read_paths[:50]]]) + if commands: + lines.append("") + lines.append("Observed commands:") + lines.extend(_render_command_lines(commands[-30:])) + if searches: + lines.extend( + [ + "", + "Observed searches:", + *[ + f"- {json.dumps(item, ensure_ascii=False, sort_keys=True)}" + for item in searches[-20:] + ], + ] + ) + if conclusion: + lines.extend(["", "Agent conclusion:", conclusion]) + if outcome: + lines.extend(["", "Task outcome:", outcome]) + lines.extend( + [ + "", + f"Provenance: repo={repo.identity}; branch={repo.branch}; head={repo.head_sha}; packet={packet_id}", + ] + ) + content = "\n".join(lines) + return bounded(content, MAX_EPISODE_CHARS), structured + + +def build_semantic_evidence(structured: dict[str, Any]) -> str: + """Format changed paths for memory extraction. + + Test and build results remain in the local evidence store for diagnostics, + but are not useful repository knowledge by default and should not steer + memory extraction toward transient verification details. + """ + modified_paths = [ + bounded(path, 500) for path in structured.get("files_modified", [])[:20] + ] + commands = structured.get("commands") or [] + if not any(command.get("status") == "failed" for command in commands): + commands = [] + + if not modified_paths and not commands: + return "" + + lines = ["Additional repository details from this session"] + if modified_paths: + lines.extend( + [ + "", + "Changed paths:", + *[f"- {path}" for path in modified_paths], + ] + ) + if commands: + lines.extend(["", "Commands run in this session:", *_render_command_lines(commands)]) + return bounded("\n".join(lines), 8000) + + +def build_extraction_messages(structured: dict[str, Any]) -> list[dict[str, str]]: + """Build the session messages sent to Mem0 for memory extraction.""" + evidence = build_semantic_evidence(structured) + messages = [ + {"role": message["role"], "content": redact(message["content"]).strip()} + for message in structured.get("extraction_messages", []) + if message.get("role") in {"user", "assistant"} and message.get("content") + ] + if evidence: + for message in reversed(messages): + if message["role"] == "assistant": + message["content"] = f"{message['content']}\n\n{evidence}" + break + else: + messages.append({"role": "assistant", "content": evidence}) + + return messages + + +def _estimated_tokens(value: str) -> int: + """Conservatively estimate tokens without adding a tokenizer dependency.""" + ascii_chars = sum(ord(char) < 128 for char in value) + return math.ceil((ascii_chars * 0.4) + (len(value) - ascii_chars)) + + +def _message_tokens(messages: list[dict[str, str]]) -> int: + return _estimated_tokens(json.dumps(messages, ensure_ascii=False)) + + +def _is_agent_assignment(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent assignment (") + + +def _is_agent_response(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent response (") + + +def extraction_message_batches( + messages: list[dict[str, str]], + *, + max_tokens: int = MAX_EXTRACTION_INPUT_TOKENS, +) -> list[list[dict[str, str]]]: + """Keep exchanges together when possible; split oversized messages to enforce the request budget.""" + if not messages or _message_tokens(messages) <= max_tokens: + return [messages] + + exchanges: list[list[dict[str, str]]] = [] + exchange: list[dict[str, str]] = [] + for message in messages: + if message.get("role") == "user" and exchange: + exchanges.append(exchange) + exchange = [] + exchange.append(message) + if exchange: + exchanges.append(exchange) + + units: list[list[dict[str, str]]] = [] + for exchange in exchanges: + if _message_tokens(exchange) <= max_tokens: + units.append(exchange) + continue + index = 0 + while index < len(exchange): + message = exchange[index] + if ( + _is_agent_assignment(message) + and index + 1 < len(exchange) + and _is_agent_response(exchange[index + 1]) + ): + units.append(exchange[index : index + 2]) + index += 2 + else: + units.append([message]) + index += 1 + + bounded_units: list[list[dict[str, str]]] = [] + for unit in units: + if _message_tokens(unit) <= max_tokens: + bounded_units.append(unit) + continue + for message in unit: + remaining = message["content"] + while remaining: + low, high = 0, len(remaining) + while low < high: + middle = (low + high + 1) // 2 + if _message_tokens([{**message, "content": remaining[:middle]}]) <= max_tokens: + low = middle + else: + high = middle - 1 + if low == 0: + raise ValueError("Extraction token budget cannot fit a message") + bounded_units.append([{**message, "content": remaining[:low]}]) + remaining = remaining[low:] + + batches: list[list[dict[str, str]]] = [] + batch: list[dict[str, str]] = [] + for unit in bounded_units: + candidate = [*batch, *unit] + if batch and _message_tokens(candidate) > max_tokens: + batches.append(batch) + batch = list(unit) + else: + batch = candidate + if batch: + batches.append(batch) + return batches + + +def _request_json( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + raw = json.dumps(payload, ensure_ascii=False).encode() + request = urllib.request.Request( + url, + data=raw, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + parsed = json.loads(response_raw or b"{}") + return parsed, len(raw), len(response_raw) + + +def _request_json_with_network_retry( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + """Retry one transient connection failure without retrying API responses.""" + try: + return _request_json(url, key, payload, timeout) + except urllib.error.HTTPError: + raise + except (urllib.error.URLError, TimeoutError, OSError): + time.sleep(0.25) + return _request_json(url, key, payload, timeout) + + +def _get_json( + url: str, key: str, timeout: float +) -> tuple[dict[str, Any] | list[Any], int]: + request = urllib.request.Request( + url, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="GET", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + return json.loads(response_raw or b"{}"), len(response_raw) + + +def _event_id(response: dict[str, Any] | list[Any]) -> str: + return str(response.get("event_id", "")) if isinstance(response, dict) else "" + + +def _stored_event_ids(value: Any) -> list[str]: + raw = str(value or "") + if not raw.startswith("["): + return [] + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return [] + return [str(item or "") for item in parsed] if isinstance(parsed, list) else [] + + +def _result_count(response: dict[str, Any] | list[Any]) -> int: + if isinstance(response, dict): + results = response.get("results") + return len(results) if isinstance(results, list) else 0 + return len(response) if isinstance(response, list) else 0 + + +def touch_handoff_heartbeat() -> None: + """Mark the worker's handoff file alive so recovery does not relaunch it.""" + path = os.environ.get("MEM0_CODE_HANDOFF_PATH", "") + if not path: + return + try: + os.utime(path) + except OSError: + pass + + +def _wait_for_event(api_url: str, key: str, event_id: str) -> tuple[str, int, int]: + """Wait for extraction to finish before a later task can search the store.""" + if not event_id: + return "MISSING", 0, 0 + wait_seconds = float(os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120")) + poll_seconds = max(float(os.environ.get("MEM0_CODE_EVENT_POLL_SECONDS", "1")), 0.1) + deadline = time.monotonic() + wait_seconds + response_chars = 0 + while time.monotonic() < deadline: + touch_handoff_heartbeat() + try: + response, size = _get_json( + f"{api_url}/v1/event/{event_id}/", + key, + min(10, poll_seconds + 5), + ) + except urllib.error.HTTPError as exc: + if exc.code not in {408, 429} and exc.code < 500: + raise + time.sleep(poll_seconds) + continue + except (urllib.error.URLError, TimeoutError, OSError): + # The extraction job is durable server-side. A transient polling + # failure must not discard a job that may still complete normally. + time.sleep(poll_seconds) + continue + response_chars += size + status = ( + str(response.get("status", "UNKNOWN")) + if isinstance(response, dict) + else "UNKNOWN" + ) + if status in {"SUCCEEDED", "FAILED"}: + return status, response_chars, _result_count(response) + time.sleep(poll_seconds) + return "TIMEOUT", response_chars, 0 + + +def _record_flush( + repo: RepoContext, + session_id: str, + reason: str, + status: str, + elapsed: float, + **extra: Any, +) -> None: + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status=status, + success=status in {"semantic-succeeded", "nothing-to-flush"}, + duration_ms=round(elapsed, 2), + **extra, + ) + + +def flush_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + key = api_key() + if not key: + telemetry.record("flush", reason=reason, status="local-only", success=False) + return {"status": "local-only", "reason": "no-api-key"} + + session_id = _session_id(hook_input) + if session_id == "unknown-session": + telemetry.record("flush", reason=reason, status="no-session-id", success=False) + return {"status": "error", "reason": "no-session-id"} + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prepared = store.prepare_flush(repo, session_id, reason) + if prepared is None: + return {"status": "nothing-to-flush"} + packet_id, events = prepared + existing_flush = store.flush_record(packet_id) or {} + + _, structured = build_episode( + repo, + session_id, + packet_id, + events, + canonical_task=bounded(hook_input.get("task", ""), 4000), + task_outcome=bounded(hook_input.get("task_outcome", ""), 2000), + ) + + metadata = {"source": _harness_source_tag} + if repo.branch and repo.branch not in {"detached", "unknown"}: + metadata["branch"] = repo.branch + if repo.head_sha: + metadata["git_sha"] = repo.head_sha + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + add_url = f"{api_url}/v3/memories/add/" + + write_user = _scope_value(user_id()) + write_project = _scope_value(repo.project_id) + if not write_user or not _scope_value(repo.app_id) or not write_project: + telemetry.record("flush", reason=reason, status="unscoped", success=False) + return {"status": "error", "reason": "wildcard-scope"} + + body = { + "agent_id": write_project, + "user_id": write_user, + "app_id": repo.app_id, + "run_id": session_id, + "metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)}, + "agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS, + "custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS, + "custom_categories": CODING_MEMORY_CATEGORIES, + "infer": True, + } + + started = time.perf_counter() + try: + stored_events = _stored_event_ids(existing_flush.get("semantic_event_id")) + existing_event = ( + "" if stored_events else str(existing_flush.get("semantic_event_id") or "") + ) + if existing_event: + existing_status, existing_resp, existing_items = _wait_for_event( + api_url, key, existing_event + ) + if existing_status == "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=existing_event, + error="", + ) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + True, + item_count=existing_items, + response_chars=existing_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=existing_items, + resumed=True, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "semantic_status": existing_status, + "memory_count": existing_items, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + if existing_status == "TIMEOUT": + elapsed = (time.perf_counter() - started) * 1000 + error = "semantic extraction event timed out" + store.update_flush(packet_id, status="semantic-timeout", error=error) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + False, + response_chars=existing_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-timeout", + elapsed, + resumed=True, + error_kind="timeout", + ) + return { + "status": "semantic-timeout", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + + message_batches = [ + batch + for batch in extraction_message_batches( + build_extraction_messages(structured) + ) + if batch + ] + batches = [(body, messages) for messages in message_batches] + if not batches: + store.update_flush(packet_id, status="semantic-succeeded", error="") + return {"status": "nothing-to-flush", "packet_id": packet_id} + operation_name = "flush-retry" if stored_events else "flush" + semantic_events = stored_events[: len(batches)] + semantic_events += [""] * (len(batches) - len(semantic_events)) + semantic_req = 0 + semantic_resp = 0 + for index, (body, messages) in enumerate(batches): + if semantic_events[index]: + continue + semantic_response, request_chars, response_chars = _request_json( + add_url, + key, + {**body, "messages": messages}, + 15, + ) + semantic_events[index] = _event_id(semantic_response) + semantic_req += request_chars + semantic_resp += response_chars + store.update_flush( + packet_id, + status="semantic-queued", + semantic_event_id=json.dumps(semantic_events), + ) + + semantic_event = semantic_events[-1] + semantic_status = "SUCCEEDED" + event_resp = 0 + semantic_items = 0 + failed_event = semantic_event + for index, queued_event in enumerate(semantic_events): + status, response_chars, item_count = _wait_for_event( + api_url, key, queued_event + ) + event_resp += response_chars + semantic_items += item_count + if status != "SUCCEEDED": + semantic_status = status + failed_event = queued_event + if status in {"FAILED", "MISSING"}: + semantic_events[index] = "" + store.update_flush( + packet_id, semantic_event_id=json.dumps(semantic_events) + ) + break + if semantic_status != "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + error = f"semantic extraction event {semantic_status.lower()}" + store.update_flush( + packet_id, status=f"semantic-{semantic_status.lower()}", error=error + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + False, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + f"semantic-{semantic_status.lower()}", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + error_kind=telemetry.error_kind(error), + ) + return { + "status": f"semantic-{semantic_status.lower()}", + "packet_id": packet_id, + "semantic_event_id": failed_event, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=json.dumps(semantic_events), + error="", + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + True, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + request_chars=semantic_req, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": semantic_event, + "semantic_status": semantic_status, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + except Exception as exc: # hooks must fail open + elapsed = (time.perf_counter() - started) * 1000 + error = bounded(str(exc), 1000) + store.update_flush(packet_id, status="error", error=error) + store.operation(repo, session_id, "flush", elapsed, False, error=error) + _record_flush( + repo, + session_id, + reason, + "error", + elapsed, + error_kind=telemetry.error_kind(exc), + ) + return {"status": "error", "packet_id": packet_id, "error": error} + + +def checkpoint_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + """Run remote extraction at a durable boundary.""" + return flush_session(store, hook_input, reason) + + + +def search_memories( + store: EvidenceStore | None, + repo: RepoContext, + session_id: str | None, + query: str, + *, + top_k: int | None = None, + category: str | None = None, + scope: str | None = None, + run_id: str | None = None, + operation: str = "search", + timeout: float = 5, +) -> MemorySearchResult: + key = api_key() + if not key or not query.strip(): + return MemorySearchResult(False, 0, 0, []) + search_once = os.environ.get( + "MEM0_CODE_SEARCH_ONCE_PER_SESSION", "false" + ).lower() in { + "1", + "true", + "yes", + "on", + } + track_session = store is not None and bool(session_id) + if ( + search_once + and track_session + and store.has_operation(repo.identity, session_id, "search") + ): + return MemorySearchResult(False, 0, 0, []) + + result_limit = min( + max( + top_k + if top_k is not None + else _int_option("top_k", "MEM0_CODE_TOP_K", 3), + 1, + ), + 20, + ) + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ValueError(f"Unknown memory category: {category}") + user, project = _scope_value(user_id()), _scope_value(repo.project_id) + if not user or not project or not _scope_value(repo.app_id): + return MemorySearchResult(False, 0, 0, []) + filters = _search_filters(user, repo, resolve_search_scope(scope)) + if category: + filters = {"AND": [filters, {"categories": {"contains": category}}]} + if run_id: + filters = {"AND": [filters, {"run_id": run_id}]} + payload = { + "query": query, + "app_id": repo.app_id, + "filters": filters, + "top_k": result_limit, + "rerank": False, + "latest_only": True, + } + url = ( + os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + + "/v3/memories/search/" + ) + started = time.perf_counter() + try: + response, request_chars, response_chars = _request_json_with_network_retry( + url, key, payload, timeout + ) + memories = ( + response if isinstance(response, list) else response.get("results", []) + ) + memories = [ + memory + for memory in memories + if isinstance(memory, dict) + and (memory.get("metadata") or {}).get("record_kind") != "task_episode" + ][:result_limit] + if track_session: + returned_memories = store.unseen(session_id, repo.identity, memories) + store.mark_injected(session_id, repo.identity, returned_memories) + already_shown_count = len(memories) - len(returned_memories) + else: + returned_memories = memories + already_shown_count = 0 + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation( + repo, + session_id, + operation, + elapsed, + True, + item_count=len(returned_memories), + request_chars=request_chars, + response_chars=response_chars, + ) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=True, + duration_ms=round(elapsed, 2), + matched_count=len(memories), + returned_count=len(returned_memories), + already_shown_count=already_shown_count, + top_k=result_limit, + has_category=bool(category), + ) + return MemorySearchResult( + succeeded=True, + matched_count=len(memories), + already_shown_count=already_shown_count, + memories=returned_memories, + ) + except Exception as exc: + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation(repo, session_id, operation, elapsed, False, error=str(exc)) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=False, + duration_ms=round(elapsed, 2), + top_k=result_limit, + has_category=bool(category), + error_kind=telemetry.error_kind(exc), + ) + return MemorySearchResult(False, 0, 0, []) + + +def format_context( + memories: list[dict[str, Any]], + heading: str = "Relevant repository memories:", +) -> str: + if not memories: + return "" + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + lines = [heading] if heading else [] + for memory in memories: + text = re.sub( + r"\s+", + " ", + redact(memory.get("memory") or memory.get("text") or ""), + ) + text = text.strip() + if not text: + continue + branch = str((memory.get("metadata") or {}).get("branch") or "").strip() + branch_label = ( + f" [learnt on branch {branch}]" + if branch.casefold() not in {"", "main", "master", "unknown", "detached"} + else "" + ) + number = len(lines) if heading else len(lines) + 1 + entry = f"{number}. {text}{branch_label}" + candidate = "\n".join([*lines, entry]) + if len(candidate) <= limit: + lines.append(entry) + continue + if not lines or (heading and len(lines) == 1): + prefix = f"{number}. " + suffix = f"…{branch_label}" + available = ( + limit + - len("\n".join(lines)) + - (1 if lines else 0) + - len(prefix) + - len(suffix) + ) + if available > 0: + lines.append(prefix + text[:available].rstrip() + suffix) + break + minimum_lines = 2 if heading else 1 + return "\n".join(lines) if len(lines) >= minimum_lines else "" + + +def format_search_result(result: MemorySearchResult) -> str: + """Return only the text the coding agent needs from an explicit memory search.""" + if not result.succeeded: + return "Memory search failed." + if result.memories: + rendered = format_context(result.memories, heading="") + if rendered: + return rendered + return "No matching memories found." + + +def combine_context(*contexts: str) -> str: + """Combine memory sources under one hard budget without repeated lines.""" + seen: set[str] = set() + lines: list[str] = [] + for context in contexts: + for line in str(context or "").splitlines(): + normalized = re.sub(r"\s+", " ", line).strip().casefold() + if not normalized or normalized in seen: + continue + seen.add(normalized) + lines.append(line.rstrip()) + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + return bounded("\n".join(lines), limit) if lines else "" + + +def _scoped_memory_ids( + api_url: str, key: str, user: str, repo: RepoContext, include_project: bool +) -> list[str]: + """List this user's memory ids for this repository, plus the shared project memory when asked.""" + ids: list[str] = [] + seen: set[str] = set() + prefix = repo.app_id + _collect_memory_ids( + api_url, key, {"user_id": user}, ids, seen, + app_id_prefix=prefix, + ) + if include_project: + for project_id in _shared_project_ids(repo): + _collect_memory_ids( + api_url, key, {"agent_id": project_id}, ids, seen, + app_id_prefix=prefix, + ) + return ids + + +def _collect_memory_ids( + api_url: str, + key: str, + filters: dict[str, Any], + ids: list[str], + seen: set[str], + *, + app_id_prefix: str = "", +) -> None: + """Page through one list filter; the list endpoint returns nothing for an OR whose user branch has no memories.""" + payload = {"filters": filters} + for page in range(1, FORGET_MAX_PAGES + 1): + parsed, _, _ = _request_json( + f"{api_url}/v2/memories/?page={page}&page_size={FORGET_PAGE_SIZE}", + key, + payload, + 15, + ) + items = parsed.get("results") if isinstance(parsed, dict) else parsed + if not isinstance(items, list) or not items: + break + for item in items: + if not isinstance(item, dict): + continue + memory_id = str(item.get("id", "")) + if not memory_id or memory_id in seen: + continue + if app_id_prefix: + item_app_id = str(item.get("app_id") or "") + if item_app_id != app_id_prefix and not item_app_id.startswith(app_id_prefix + "/"): + continue + seen.add(memory_id) + ids.append(memory_id) + if len(items) < FORGET_PAGE_SIZE: + break + + +def _delete_memory(api_url: str, key: str, memory_id: str) -> bool: + request = urllib.request.Request( + f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/", + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="DELETE", + ) + try: + with urllib.request.urlopen(request, timeout=15): + return True + except Exception: + return False + + +def forget_remote_repo( + repo: RepoContext, *, include_project_memory: bool = False +) -> dict[str, Any]: + """Delete this user's memories for this repository; project memory is shared, so only on request.""" + key = api_key() + if not key: + telemetry.record("forget", repo=repo, success=False, error_kind="no-api-key") + return {"status": "error", "error": "Mem0 API key is not configured"} + user = _scope_value(user_id()) + if not user or not _scope_value(repo.app_id) or not _scope_value(repo.project_id): + telemetry.record("forget", repo=repo, success=False, error_kind="unscoped") + return { + "status": "error", + "error": "Refusing to forget: the user or repository scope is a wildcard", + } + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + try: + memory_ids = _scoped_memory_ids(api_url, key, user, repo, include_project_memory) + except Exception as exc: + telemetry.record( + "forget", repo=repo, success=False, error_kind=telemetry.error_kind(exc) + ) + return {"status": "error", "error": bounded(str(exc), 1000)} + deleted = sum(_delete_memory(api_url, key, memory_id) for memory_id in memory_ids) + failed = len(memory_ids) - deleted + telemetry.record("forget", repo=repo, success=not failed, item_count=deleted) + if failed: + return { + "status": "partial", + "deleted": deleted, + "failed": failed, + "error": f"{failed} of {len(memory_ids)} memories could not be deleted", + } + return {"status": "deleted", "deleted": deleted} + + +def _doctor_mem0_authentication(repo: RepoContext) -> dict[str, Any]: + """Verify the configured key with one read-only, repository-scoped search.""" + key = api_key() + if not key: + return {"ok": False, "detail": "API key missing"} + payload = { + "query": "Mem0 authentication check", + "filters": { + "AND": [ + {"user_id": user_id()}, + {"app_id": repo.app_id}, + ] + }, + "top_k": 1, + "threshold": 1.0, + "rerank": False, + } + url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + started = time.perf_counter() + try: + _request_json(f"{url}/v3/memories/search/", key, payload, 5) + except Exception as exc: + return {"ok": False, "detail": bounded(str(exc), 300)} + elapsed = (time.perf_counter() - started) * 1000 + return {"ok": True, "detail": f"connected ({elapsed:.0f} ms)"} + + +def _doctor_user_id() -> dict[str, Any]: + """Flag a configured user ID the plugin refuses, since the silent fallback surprises people.""" + configured = _plugin_option("user_id", "MEM0_CODE_USER_ID") or os.environ.get( + "MEM0_USER_ID", "" + ) + if configured and not _scope_value(configured): + return { + "ok": False, + "detail": f"configured user_id {configured!r} is a wildcard; using {user_id()!r}", + } + return {"ok": True, "detail": user_id()} + + +def doctor(cwd: str | None = None) -> dict[str, Any]: + repo = resolve_repo(cwd) + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + checks: dict[str, dict[str, Any]] = { + "python": { + "ok": tuple(sys.version_info[:2]) >= (3, 10), + "detail": f"{sys.version_info.major}.{sys.version_info.minor}", + }, + "data_directory": { + "ok": os.access(directory, os.W_OK), + "detail": str(directory), + }, + "mem0_api_key": { + "ok": bool(api_key()), + "detail": "configured" if api_key() else "missing", + }, + "repository": { + "ok": bool(repo.identity), + "detail": repo.identity, + }, + "user_id": _doctor_user_id(), + "mem0_authentication": _doctor_mem0_authentication(repo), + } + return { + "ok": all(bool(value["ok"]) for value in checks.values()), + "plugin_version": PLUGIN_VERSION, + "repo_id": repo.identity, + "app_id": repo.app_id, + "user_id": user_id(), + "checks": checks, + } diff --git a/integrations/cursor-plugin/core/telemetry.py b/integrations/cursor-plugin/core/telemetry.py new file mode 100644 index 000000000..249595475 --- /dev/null +++ b/integrations/cursor-plugin/core/telemetry.py @@ -0,0 +1,397 @@ +#!/usr/bin/env python3 +"""Anonymous usage telemetry for Mem0 agent plugins. + +Hooks run on a 3-6 second budget and fire on every tool call, so recording never +touches the network: `record` appends one JSON line to a local spool and returns. +A detached `python3 telemetry.py` drains the spool in one batched PostHog request, +started once per session and again from the flush worker that is already detached. + +Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false. + +Never sends prompts, memory text, queries, file paths, repository names, or API +keys: only event names, durations, counts, coarse outcomes, and salted hashes. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import platform +import subprocess +import sys +import time +import urllib.error +import urllib.request +import uuid +from pathlib import Path +from typing import Any + +import memory_core + +_harness: str = "generic" +_source_tag: str = "MEM0_PLUGIN" +_PRIVATE_KEYS = { + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +} + + +def init(harness: str = "generic", source_tag: str = "") -> None: + global _harness, _source_tag + _harness = harness + _source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN" + +POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" +POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" +POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/" +EVENT_PREFIX = "code" +SPOOL_LIMIT_BYTES = 256 * 1024 +BATCH_SIZE = 100 +SEND_TIMEOUT = 5 +CLAIM_STALE_SECONDS = 120 +CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60 + + +def is_enabled() -> bool: + """Whether telemetry is switched on for this process.""" + return os.environ.get("MEM0_TELEMETRY", "true").strip().lower() not in { + "false", + "0", + "no", + "off", + } + + +def _digest(value: str, length: int = 16) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length] + + +def _safe_value(value: Any) -> Any: + if isinstance(value, str): + return memory_core.redact(value) + if isinstance(value, dict): + return { + key: _safe_value(item) + for key, item in value.items() + if "".join(character for character in str(key).lower() if character.isalnum()) + not in _PRIVATE_KEYS + } + if isinstance(value, (list, tuple)): + return [_safe_value(item) for item in value] + if value is None or isinstance(value, (bool, int, float)): + return value + return memory_core.redact(value) + + +def _spool_path() -> Path: + return memory_core.data_dir() / "telemetry.jsonl" + + +def _identity_path() -> Path: + return memory_core.data_dir() / "telemetry-identity.json" + + +def _read_identity() -> dict[str, str]: + try: + value = json.loads(_identity_path().read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + return value if isinstance(value, dict) else {} + + +def _write_identity(identity: dict[str, str]) -> None: + path = _identity_path() + temporary = path.with_suffix(f".{os.getpid()}.tmp") + try: + path.parent.mkdir(parents=True, exist_ok=True) + temporary.write_text(json.dumps(identity), encoding="utf-8") + temporary.replace(path) + except OSError: + try: + temporary.unlink() + except OSError: + pass + + +def anonymous_id(identity: dict[str, str] | None = None) -> str: + """Per-machine anonymous identifier, created and persisted on first use.""" + identity = _read_identity() if identity is None else identity + existing = identity.get("anonymous_id") + if existing: + return existing + created = f"code-anon-{uuid.uuid4().hex}" + identity["anonymous_id"] = created + _write_identity(identity) + return created + + +def is_first_run() -> bool: + """Whether this machine has never recorded a plugin event before.""" + return not _identity_path().exists() + + +def record( + event: str, + *, + repo: Any = None, + session_id: str | None = None, + **properties: Any, +) -> None: + """Append one event to the local spool. Never blocks and never raises.""" + if not is_enabled(): + return + try: + spool = _spool_path() + try: + if spool.stat().st_size > SPOOL_LIMIT_BYTES: + return + except OSError: + pass + properties = _safe_value(properties) + properties.update( + harness=_harness, + plugin_version=memory_core.PLUGIN_VERSION, + os=sys.platform, + python_version=platform.python_version(), + ) + if repo is not None: + properties["repo_hash"] = _digest(getattr(repo, "identity", "")) + if session_id: + properties["session_hash"] = _digest(session_id) + line = json.dumps( + { + "event": f"{EVENT_PREFIX}.{event}", + "timestamp": memory_core.utc_now(), + "properties": { + key: value for key, value in properties.items() if value is not None + }, + }, + separators=(",", ":"), + default=str, + ) + spool.parent.mkdir(parents=True, exist_ok=True) + with spool.open("a", encoding="utf-8") as handle: + handle.write(line + "\n") + except Exception: + pass + + +def error_kind(exc: BaseException | str) -> str: + """Coarse, content-free label for a failure, safe to send.""" + text = exc if isinstance(exc, str) else f"{type(exc).__name__}: {exc}" + lowered = text.lower() + if "timed out" in lowered or "timeout" in lowered: + return "timeout" + if "401" in lowered or "403" in lowered or "unauthor" in lowered or "forbidden" in lowered: + return "auth" + if "429" in lowered or "rate limit" in lowered: + return "rate-limited" + if any(code in lowered for code in ("500", "502", "503", "504")): + return "server-error" + if "400" in lowered or "422" in lowered: + return "bad-request" + if isinstance(exc, str): + return "other" + if isinstance(exc, urllib.error.URLError): + return "network" + return type(exc).__name__ + + +def spawn_flush() -> bool: + """Start the detached sender that drains the spool.""" + if not is_enabled(): + return False + try: + if not _spool_path().exists() and not any( + memory_core.data_dir().glob("telemetry-*.sending") + ): + return False + subprocess.Popen( + [sys.executable, str(Path(__file__).resolve())], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + close_fds=True, + **memory_core.detached_process_kwargs(), + ) + return True + except Exception: + return False + + +def _claim_spool() -> Path | None: + """Rename the spool aside so exactly one sender owns each batch.""" + directory = memory_core.data_dir() + claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending" + spool = _spool_path() + try: + spool.replace(claim) + return claim + except OSError: + pass + now = time.time() + for orphan in sorted(directory.glob("telemetry-*.sending")): + try: + age = now - orphan.stat().st_mtime + except OSError: + continue + if age > CLAIM_EXPIRY_SECONDS: + try: + orphan.unlink() + except OSError: + pass + continue + if age < CLAIM_STALE_SECONDS: + continue + try: + orphan.replace(claim) + return claim + except OSError: + continue + return None + + +def _resolve_email(key: str) -> str: + """Trade the API key for the account email so events join other Mem0 surfaces.""" + url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/" + request = urllib.request.Request( + url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json"} + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT) as response: + payload = json.loads(response.read().decode("utf-8")) + except Exception: + return "" + email = payload.get("user_email") if isinstance(payload, dict) else "" + return email if isinstance(email, str) else "" + + +def _post(payload: dict[str, Any], url: str) -> bool: + request = urllib.request.Request( + url, + data=json.dumps(payload, default=str).encode("utf-8"), + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT): + return True + except Exception: + return False + + +def resolve_distinct_id() -> tuple[str, str]: + """Return the PostHog distinct id and the anonymous id it replaced, if any.""" + identity = _read_identity() + email = identity.get("email", "") + if email: + return email, "" + key = memory_core.api_key() + if not key: + return anonymous_id(identity), "" + email = _resolve_email(key) + if not email: + return anonymous_id(identity), "" + previous = identity.get("anonymous_id", "") + identity["email"] = email + _write_identity(identity) + return email, previous + + +def flush() -> int: + """Drain claimed spools to PostHog and return the number of events sent.""" + if not is_enabled(): + return 0 + claim = _claim_spool() + if claim is None: + return 0 + try: + lines = claim.read_text(encoding="utf-8").splitlines() + except OSError: + return 0 + events = [] + for line in lines: + try: + value = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(value, dict) and value.get("event"): + events.append(value) + if not events: + try: + claim.unlink() + except OSError: + pass + return 0 + + distinct_id, aliased_anonymous_id = resolve_distinct_id() + if aliased_anonymous_id: + _post( + { + "api_key": POSTHOG_API_KEY, + "event": "$identify", + "distinct_id": distinct_id, + "properties": { + "$anon_distinct_id": aliased_anonymous_id, + "$lib": "posthog-python", + }, + }, + POSTHOG_CAPTURE_URL, + ) + + sent = 0 + for start in range(0, len(events), BATCH_SIZE): + batch = [ + { + "event": event["event"], + "distinct_id": distinct_id, + "timestamp": event.get("timestamp"), + "properties": { + "source": _source_tag, + "language": "python", + "$process_person_profile": False, + "$lib": "posthog-python", + **(event.get("properties") or {}), + }, + } + for event in events[start : start + BATCH_SIZE] + ] + if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL): + return sent + sent += len(batch) + try: + claim.unlink() + except OSError: + pass + return sent + + +def main() -> int: + flush() + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception: + raise SystemExit(0) diff --git a/integrations/cursor-plugin/hooks/adapter.py b/integrations/cursor-plugin/hooks/adapter.py new file mode 100644 index 000000000..d167f9a28 --- /dev/null +++ b/integrations/cursor-plugin/hooks/adapter.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Translate Cursor's native hooks into the shared Mem0 runtime.""" + +from __future__ import annotations + +import contextlib +import io +import json +import os +import sys +from pathlib import Path + +HERE = Path(__file__).resolve() +BUNDLED_CORE = HERE.parent.parent / "core" +CORE = BUNDLED_CORE if BUNDLED_CORE.is_dir() else HERE.parents[2] / "core" / "python" +sys.path.insert(0, str(CORE)) + +import hook_runner # noqa: E402 +import telemetry # noqa: E402 +from memory_core import ( # noqa: E402 + configure_harness, + record_sidekick_start, + record_sidekick_stop, + record_tool, +) + +EVENTS = { + "sessionStart": "session-start", + "beforeSubmitPrompt": "user-prompt", + "postToolUse": "post-tool", + "postToolUseFailure": "post-tool-failure", + "afterAgentResponse": "assistant-stop", + "subagentStart": "sidekick-start", + "subagentStop": "sidekick-stop", + "stop": "stop", + "sessionEnd": "session-end", + "preCompact": "pre-compact", +} + + +def normalize(payload: dict, event: str) -> dict: + value = dict(payload) + value.setdefault("session_id", value.get("conversation_id") or value.get("parent_conversation_id", "")) + roots = value.get("workspace_roots") or [] + if roots: + value.setdefault("cwd", roots[0]) + if "tool_output" in value: + value.setdefault("tool_response", value["tool_output"]) + if "error_message" in value: + value.setdefault("tool_response", value["error_message"]) + if "text" in value: + value.setdefault("last_assistant_message", value["text"]) + if "summary" in value: + value.setdefault("last_assistant_message", value["summary"]) + if "subagent_id" in value: + value.setdefault("agent_id", value["subagent_id"]) + if "subagent_type" in value: + value.setdefault("agent_type", value["subagent_type"]) + return {"action": EVENTS[event], "payload": value} + + +def _record_failure(store, payload): + return record_tool(store, payload, failed=True) + + +def _record_response(store, payload): + hook_runner.default_record_stop(store, payload) + + +def _record_sidekick_start(store, payload): + record_sidekick_start(store, payload, inject_context=False) + return {"permission": "allow"} + + +def _record_sidekick_stop(store, payload): + record_sidekick_stop(store, payload) + + +def main() -> int: + if len(sys.argv) != 2 or sys.argv[1] not in EVENTS: + return 2 + event = sys.argv[1] + try: + raw = json.load(sys.stdin) + except (json.JSONDecodeError, OSError): + raw = {} + normalized = normalize(raw if isinstance(raw, dict) else {}, event) + action = normalized["action"] + arguments = { + "session-end": ["flush", "--reason", "session-end"], + "pre-compact": ["flush", "--reason", "pre-compact"], + }.get(action, [action]) + sys.argv = [sys.argv[0], *arguments] + sys.stdin = io.StringIO(json.dumps(normalized["payload"])) + + configure_harness("cursor", data_dir_name="cursor-plugin", source_tag="cursor_plugin") + telemetry.init(harness="cursor", source_tag="CURSOR_PLUGIN") + with contextlib.redirect_stdout(io.StringIO()): + result = hook_runner.run( + extra_actions={ + "post-tool-failure": _record_failure, + "assistant-stop": _record_response, + "sidekick-start": _record_sidekick_start, + "sidekick-stop": _record_sidekick_stop, + }, + automatic_flush_reasons={"session-end", "pre-compact"}, + ) + if event == "sessionStart": + environment = { + key: value + for key in ( + "PLUGIN_OPTION_API_KEY", + "PLUGIN_OPTION_USER_ID", + "PLUGIN_OPTION_TOP_K", + "PLUGIN_OPTION_MAX_CONTEXT_CHARS", + "PLUGIN_OPTION_SEARCH_SCOPE", + ) + if (value := os.environ.get(key)) + } + if environment: + print(json.dumps({"env": environment})) + elif event == "subagentStart": + print(json.dumps({"permission": "allow"})) + return result + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception as exc: + hook_runner.log_failure(exc) + raise SystemExit(0) from None diff --git a/integrations/cursor-plugin/hooks/hooks.json b/integrations/cursor-plugin/hooks/hooks.json new file mode 100644 index 000000000..1e4875d50 --- /dev/null +++ b/integrations/cursor-plugin/hooks/hooks.json @@ -0,0 +1,15 @@ +{ + "version": 1, + "hooks": { + "sessionStart": [{ "command": "PLUGIN_OPTION_API_KEY=\"${api_key}\" PLUGIN_OPTION_USER_ID=\"${user_id}\" PLUGIN_OPTION_TOP_K=\"${top_k}\" PLUGIN_OPTION_MAX_CONTEXT_CHARS=\"${max_context_chars}\" PLUGIN_OPTION_SEARCH_SCOPE=\"${search_scope}\" python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" sessionStart", "timeout": 5 }], + "beforeSubmitPrompt": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" beforeSubmitPrompt", "timeout": 6 }], + "postToolUse": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" postToolUse", "timeout": 3 }], + "postToolUseFailure": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" postToolUseFailure", "timeout": 3 }], + "afterAgentResponse": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" afterAgentResponse", "timeout": 3 }], + "subagentStart": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" subagentStart", "matcher": "^sidekick$", "timeout": 5 }], + "subagentStop": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" subagentStop", "matcher": "^sidekick$", "timeout": 5 }], + "stop": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" stop", "timeout": 3 }], + "preCompact": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" preCompact", "timeout": 5 }], + "sessionEnd": [{ "command": "python3 \"${CURSOR_PLUGIN_ROOT}/hooks/adapter.py\" sessionEnd", "timeout": 5 }] + } +} diff --git a/integrations/cursor-plugin/mcp.json b/integrations/cursor-plugin/mcp.json new file mode 100644 index 000000000..b17c1186c --- /dev/null +++ b/integrations/cursor-plugin/mcp.json @@ -0,0 +1,18 @@ +{ + "mcpServers": { + "mem0": { + "type": "stdio", + "command": "python3", + "args": [ + "${CURSOR_PLUGIN_ROOT}/core/mcp_server.py" + ], + "env": { + "PLUGIN_OPTION_API_KEY": "${api_key}", + "PLUGIN_OPTION_USER_ID": "${user_id}", + "PLUGIN_OPTION_TOP_K": "${top_k}", + "PLUGIN_OPTION_MAX_CONTEXT_CHARS": "${max_context_chars}", + "PLUGIN_OPTION_SEARCH_SCOPE": "${search_scope}" + } + } + } +} diff --git a/integrations/cursor-plugin/plugin-build.json b/integrations/cursor-plugin/plugin-build.json new file mode 100644 index 000000000..bddfbec4a --- /dev/null +++ b/integrations/cursor-plugin/plugin-build.json @@ -0,0 +1,15 @@ +{ + "id": "mem0", + "version": "0.3.1", + "homepage": "https://docs.mem0.ai/integrations/cursor", + "native": { + "pluginRoot": "${CURSOR_PLUGIN_ROOT}", + "files": { + ".cursor-plugin/plugin.json": ".cursor-plugin/plugin.json", + "hooks/hooks.json": "hooks/hooks.json", + "hooks/adapter.py": "hooks/adapter.py", + "agents/sidekick.md": "agents/sidekick.md", + "mcp.json": "mcp.json" + } + } +} diff --git a/integrations/cursor-plugin/skills/forget/SKILL.md b/integrations/cursor-plugin/skills/forget/SKILL.md new file mode 100644 index 000000000..c3af4d3df --- /dev/null +++ b/integrations/cursor-plugin/skills/forget/SKILL.md @@ -0,0 +1,26 @@ +--- +name: forget +description: Delete the Mem0 memories stored for this repository and this user. Use when the user asks to forget, clear, wipe, or delete memories. +disable-model-invocation: true +--- + +# Forget this repository's memories + +This permanently deletes remote memories. Before running anything, tell the +user exactly what will be deleted: their own memories for this repository +only. The repository's project memory is shared by everyone who works in it, +so it stays unless the user explicitly asks to delete that too. + +After the user confirms, run: + +```bash +python3 "${CURSOR_PLUGIN_ROOT}/core/memory_cli.py" --harness "cursor" forget --remote --yes +``` + +If the user also asked to delete the repository's shared project memory, add +`--include-project-memory` and say that this removes it for every teammate. + +Report what the command output says was deleted. If the user only wants local +data cleared (evidence log, pending queue), run the same command without +`--remote`. Never pass `--yes` before the user has confirmed in this +conversation. diff --git a/integrations/cursor-plugin/skills/pause/SKILL.md b/integrations/cursor-plugin/skills/pause/SKILL.md new file mode 100644 index 000000000..634385145 --- /dev/null +++ b/integrations/cursor-plugin/skills/pause/SKILL.md @@ -0,0 +1,20 @@ +--- +name: pause +description: Pause Mem0 memory capture on this machine. Use when the user wants to stop memories being recorded, for example for private work or experiments. +disable-model-invocation: true +--- + +# Pause memory capture + +To pause (hooks stop capturing and sending session content; a minimal +anonymous telemetry ping still fires at session start unless +`MEM0_TELEMETRY=false`): + +```bash +python3 "${CURSOR_PLUGIN_ROOT}/core/memory_cli.py" --harness "cursor" pause +``` + +Confirm the new state back to the user, and remind them that already-created +memories still exist and remain searchable. Pending unsent packets are held +while paused, not expired, and are delivered after resuming. To turn capture +back on, use `/mem0:resume`. diff --git a/integrations/cursor-plugin/skills/remember/SKILL.md b/integrations/cursor-plugin/skills/remember/SKILL.md new file mode 100644 index 000000000..7be5fc9be --- /dev/null +++ b/integrations/cursor-plugin/skills/remember/SKILL.md @@ -0,0 +1,21 @@ +--- +name: remember +description: Acknowledge a "remember this" request and make sure it is captured well. Use when the user explicitly asks to remember, note, or save something for future sessions. +disable-model-invocation: true +--- + +# Remember something for future sessions + +Mem0 creates memories from the session automatically — there is no separate +write command. When the user asks to remember something: + +1. Restate the fact clearly and completely in your reply, in one or two + sentences, including any names, values, or paths it depends on. Your visible + reply is what memory extraction reads, so a precise restatement is what gets + remembered. +2. Tell the user it will be saved with this session's memories when the session + ends or compacts, and that it will surface in future sessions in this + repository (they can check later with /mem0:search). + +Do not invent a storage confirmation or a memory ID — creation happens in the +background after the session. diff --git a/integrations/cursor-plugin/skills/resume/SKILL.md b/integrations/cursor-plugin/skills/resume/SKILL.md new file mode 100644 index 000000000..d45bd545d --- /dev/null +++ b/integrations/cursor-plugin/skills/resume/SKILL.md @@ -0,0 +1,19 @@ +--- +name: resume +description: Resume Mem0 memory capture after it was paused with /mem0:pause. +disable-model-invocation: true +--- + +# Resume memory capture + +Resume memory capture for this machine. + +Run: + +```bash +python3 "${CURSOR_PLUGIN_ROOT}/core/memory_cli.py" --harness "cursor" resume +``` + +Confirm to the user that capture is active again. New sessions record evidence and +create memories as normal; nothing that happened while paused is retroactively +captured. diff --git a/integrations/cursor-plugin/skills/search/SKILL.md b/integrations/cursor-plugin/skills/search/SKILL.md new file mode 100644 index 000000000..7a7ddfa90 --- /dev/null +++ b/integrations/cursor-plugin/skills/search/SKILL.md @@ -0,0 +1,28 @@ +--- +name: search +description: Search memories from earlier Cursor sessions in this repository. Use it when earlier work may already explain the code, error, decision, or command you need, so you can avoid repeating file reads, searches, or experiments. +argument-hint: "[question] [--top-k number] [--category category-name] [--scope repo|dir|mine] [--run-id session-id]" +disable-model-invocation: true +--- + +# Search memories + +Call `search_memories` with the user's question. Treat `--top-k`, `--category`, +`--scope`, and `--run-id` as tool arguments instead of including them in the +query. + +Omit `top_k` to use Mem0's configured default. Omit `category` to search every +category; a category is a best-effort label Mem0 assigned when it saved the +memory, so if a category search misses, repeat it without the category. Omit +`scope` to use the configured default, normally `repo`: this repository's +shared memory, which everyone who works in it contributes to, plus your own +preferences. + +Pass `scope` when the question needs something else: `dir` to narrow the +shared memory to the directory you are working in (a package inside a +monorepo), `mine` for your own preferences alone. + +Pass `run_id` with any scope to retrieve memories saved in a specific coding-agent +session. Omit `run_id` to search across sessions. It filters the memories returned; +it does not identify the session making the search request. Use a known session ID, +never invent one. Return the tool's result directly. diff --git a/integrations/cursor-plugin/skills/status/SKILL.md b/integrations/cursor-plugin/skills/status/SKILL.md new file mode 100644 index 000000000..23ad94e33 --- /dev/null +++ b/integrations/cursor-plugin/skills/status/SKILL.md @@ -0,0 +1,23 @@ +--- +name: status +description: Show whether Mem0 memory is working in this repository, covering configuration, capture state, pending flushes, and whether the Mem0 API key is valid. Use when the user asks whether memory is on, why a memory is missing, or anything looks broken. +disable-model-invocation: false +--- + +# Memory status + +Run both commands and report the combined result in plain language: + +```bash +python3 "${CURSOR_PLUGIN_ROOT}/core/memory_cli.py" --harness "cursor" status --json +python3 "${CURSOR_PLUGIN_ROOT}/core/memory_cli.py" --harness "cursor" doctor +``` + +Summarize, using only fields the JSON actually reports: whether capture is +active or paused, the user ID and repository scope (`repo_id`), whether an +API key is configured, the event/flush/retrieval counts (`flushes` is the +number of completed flushes, not a pending count), and the doctor check +results. If doctor reports an authentication failure (401 / invalid key), say +clearly that the Mem0 API key is invalid or expired and that memories are NOT +being created. Never report an auth failure as "no memories found". Suggest +reinstalling with `--config api_key=...` in that case. diff --git a/integrations/cursor-plugin/tests/test_adapter.py b/integrations/cursor-plugin/tests/test_adapter.py new file mode 100644 index 000000000..d0956e410 --- /dev/null +++ b/integrations/cursor-plugin/tests/test_adapter.py @@ -0,0 +1,112 @@ +from __future__ import annotations + +import importlib.util +import json +import sys +from pathlib import Path + +import pytest + +HOST = Path(__file__).resolve().parents[1] +CORE_ROOT = HOST.parent / "agent-plugin-core" +sys.path.insert(0, str(CORE_ROOT)) +from build.build import build # noqa: E402 + +SPEC = importlib.util.spec_from_file_location("cursor_adapter", HOST / "hooks" / "adapter.py") +assert SPEC and SPEC.loader +adapter = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(adapter) + + +@pytest.mark.parametrize( + ("event", "action"), + [ + ("sessionStart", "session-start"), + ("beforeSubmitPrompt", "user-prompt"), + ("postToolUse", "post-tool"), + ("postToolUseFailure", "post-tool-failure"), + ("afterAgentResponse", "assistant-stop"), + ("subagentStart", "sidekick-start"), + ("subagentStop", "sidekick-stop"), + ("stop", "stop"), + ("sessionEnd", "session-end"), + ("preCompact", "pre-compact"), + ], +) +def test_normalizes_cursor_events(event: str, action: str) -> None: + normalized = adapter.normalize( + { + "conversation_id": "conversation-1", + "workspace_roots": ["/worktree"], + "tool_output": "done", + "text": "assistant response", + }, + event, + ) + + assert normalized["action"] == action + assert normalized["payload"]["session_id"] == "conversation-1" + assert normalized["payload"]["cwd"] == "/worktree" + assert normalized["payload"]["tool_response"] == "done" + assert normalized["payload"]["last_assistant_message"] == "assistant response" + + +def test_after_agent_response_does_not_return_internal_context(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(adapter.hook_runner, "default_record_stop", lambda store, payload: (object(), "session")) + + assert adapter._record_response(object(), {}) is None + + +def test_normalizes_cursor_sidekick_fields() -> None: + started = adapter.normalize( + { + "subagent_id": "agent-1", + "subagent_type": "sidekick", + "parent_conversation_id": "parent-1", + }, + "subagentStart", + )["payload"] + stopped = adapter.normalize({"summary": "done"}, "subagentStop")["payload"] + + assert started["agent_id"] == "agent-1" + assert started["agent_type"] == "sidekick" + assert started["session_id"] == "parent-1" + assert stopped["last_assistant_message"] == "done" + + +def test_cursor_sidekick_records_lifecycle_without_blocking(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(adapter, "record_sidekick_start", lambda store, payload, *, inject_context: "" if not inject_context else pytest.fail("cannot inject context")) + stopped = [] + monkeypatch.setattr(adapter, "record_sidekick_stop", lambda store, payload: stopped.append(payload)) + + assert adapter._record_sidekick_start(object(), {}) == {"permission": "allow"} + assert adapter._record_sidekick_stop(object(), {"summary": "done"}) is None + assert stopped == [{"summary": "done"}] + + +def test_cursor_hooks_use_native_flat_entries() -> None: + hooks = json.loads((HOST / "hooks" / "hooks.json").read_text(encoding="utf-8")) + + assert hooks["version"] == 1 + assert set(hooks["hooks"]) >= { + "sessionStart", + "beforeSubmitPrompt", + "postToolUse", + "subagentStart", + "subagentStop", + "stop", + "sessionEnd", + } + assert hooks["hooks"]["subagentStart"][0]["matcher"] == "^sidekick$" + assert hooks["hooks"]["subagentStop"][0]["matcher"] == "^sidekick$" + assert all("hooks" not in entry for entries in hooks["hooks"].values() for entry in entries) + + +def test_native_cursor_bundle_is_self_contained(tmp_path: Path) -> None: + root = build("cursor", "native", tmp_path / "cursor") + + manifest = json.loads((root / ".cursor-plugin" / "plugin.json").read_text(encoding="utf-8")) + assert manifest["hooks"] == "./hooks/hooks.json" + assert (root / "hooks" / "adapter.py").is_file() + assert (root / "agents" / "sidekick.md").is_file() + assert not any(path.is_symlink() for path in root.rglob("*")) diff --git a/integrations/deepseek-plugin/README.md b/integrations/deepseek-plugin/README.md index 088835d08..fc6617c3f 100644 --- a/integrations/deepseek-plugin/README.md +++ b/integrations/deepseek-plugin/README.md @@ -2,10 +2,12 @@ [Mem0](https://mem0.ai) long-term memory as a native [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness) (Cordis) plugin. -It gives a Harness agent two memory tools backed by the Mem0 SDK, so recall and writes persist across sessions: +It gives a Harness agent automatic long-term memory plus two explicit memory tools backed by the Mem0 SDK: -| Tool | Does | +| Capability | Does | |---|---| +| Auto-recall | Searches Mem0 for the latest human prompt and adds unseen results to the model context | +| Auto-capture | Stores the human/assistant messages from each completed turn | | `search_memory` | Recall facts from Mem0 relevant to a query | | `add_memory` | Store a fact in Mem0 for future sessions | @@ -13,33 +15,47 @@ Unlike the local/file-based memory plugins in the ecosystem, Mem0 is a managed b ## How it works -A Cordis plugin is a module exporting `apply(ctx, config)`. This one declares `inject = ['tools']` so it waits for the harness tool registry, then registers the two tools via `ctx.tools.register(defineTool(...))`. When the plugin unmounts, the tools are removed automatically (Cordis revertible effects). +A Cordis plugin is a module exporting `apply(ctx, config)`. This one waits for the Harness tool and system-prompt services, then uses the native extension points: + +- `system-prompt/assemble` recalls memory before a model request. +- `session/event` captures only completed turns from the durable event stream. +- `ctx.tools.register(...)` exposes explicit search and add tools. + +Cordis owns listener and tool cleanup when the plugin unmounts. Every automatic path is fail-open: a memory API failure does not block the agent. ``` [ mem0ai SDK ] <-- managed memory, owned by Mem0 | -[ deepseek-plugin: apply(ctx) -> ctx.tools.register(...) ] <-- this package +[ deepseek-plugin: prompt + session listeners, memory tools ] <-- this package | [ DeepSeek Harness ] <-- the agent, loaded via cordis.yml ``` ## Try it locally -1. Build the plugin: +1. Build and pack the plugin: ```sh cd integrations/deepseek-plugin - pnpm install + pnpm install --frozen-lockfile pnpm build + mkdir -p /tmp/mem0-deepseek-plugin + pnpm pack --pack-destination /tmp/mem0-deepseek-plugin ``` 2. Set your Mem0 key: ```sh export MEM0_API_KEY=... ``` -3. Point Harness at it. Copy `cordis.example.yml`, set the absolute path to `dist/index.js` and your `userId`, then: +3. Install it into a disposable Harness profile: ```sh - pnpm dsh web --patch ./integrations/deepseek-plugin/cordis.example.yml + DSH_HOME=/tmp/mem0-dsh-dev pnpm dlx @deepseek-ai/dsh@0.1.1-rc.2 \ + plugin --profile headless add /tmp/mem0-deepseek-plugin/mem0-deepseek-plugin-0.3.0.tgz ``` -4. Open http://127.0.0.1:3080 and ask the agent to remember something, then recall it in a later turn. +4. Copy `cordis.example.yml`, set its installed package path and your `userId`, then run Harness with the same profile: + ```sh + DSH_HOME=/tmp/mem0-dsh-dev pnpm dlx @deepseek-ai/dsh@0.1.1-rc.2 \ + web --patch ./integrations/deepseek-plugin/cordis.example.yml + ``` +5. Open http://127.0.0.1:3080 and ask the agent to remember something, then recall it in a later turn. For a Mem0 Platform on-prem or dedicated deployment, point `config.host` at that base URL (defaults to `api.mem0.ai`). `host` is a Platform base-URL override — it is not a switch to self-hosted Mem0 OSS, whose server exposes a different API surface. @@ -49,7 +65,10 @@ For a Mem0 Platform on-prem or dedicated deployment, point `config.host` at that |---|---|---|---| | `apiKey` | no | `$MEM0_API_KEY` | Mem0 platform API key | | `userId` | yes | | Entity that owns the memories | +| `allowUserOverride` | no | `false` | Permit model-selected access to a different user only in a trusted multi-user deployment | | `host` | no | `api.mem0.ai` | Platform base URL (on-prem / dedicated) | +| `autoRecall` | no | `true` | Recall relevant memory before model requests | +| `autoCapture` | no | `true` | Store completed human/assistant turns | ## Telemetry @@ -59,4 +78,6 @@ The plugin also sends anonymous usage events (which tool ran, duration, result c ## Status -Developer preview. Tracks the DeepSeek Harness v0.1 plugin API (`@deepseek-ai/cordis`, `@deepseek-ai/dsh-tools`), which is young and moving; pin versions once it stabilizes. Auto-capture (store turns without an explicit tool call) and auto-recall (inject memory into the prompt at assembly) are planned once the harness session/assembly event API is confirmed. +Developer preview. Tracks the DeepSeek Harness v0.1 plugin API, which is young and moving. Harness capability packages are peer dependencies supplied by the host; this package pins matching release-candidate versions for local typechecking and tests. + +Per-call `userId` overrides are rejected unless the operator enables `allowUserOverride: true`. Automatic recall and capture always use the configured user. diff --git a/integrations/deepseek-plugin/cordis.example.yml b/integrations/deepseek-plugin/cordis.example.yml index 7a0ac3bb9..5ed84b162 100644 --- a/integrations/deepseek-plugin/cordis.example.yml +++ b/integrations/deepseek-plugin/cordis.example.yml @@ -1,19 +1,17 @@ # Example DeepSeek Harness config that loads deepseek-plugin alongside the built-in # tools plugin. Load with: # -# pnpm dsh web --patch ./integrations/deepseek-plugin/cordis.example.yml +# DSH_HOME=/tmp/mem0-dsh-dev pnpm dlx @deepseek-ai/dsh@0.1.1-rc.2 \ +# web --patch ./integrations/deepseek-plugin/cordis.example.yml # -# The tools plugin (and its system-prompt dependency) must be present, because -# the memory tools contribute schemas the system prompt renders. - -- name: "@deepseek-ai/dsh-system-prompt" -- name: "@deepseek-ai/dsh-tools" - insert: - id: mem0 - # Absolute path to the built plugin (run `pnpm build` first), or point - # at src/index.ts when running through tsx during development. - name: "/absolute/path/to/mem0/integrations/deepseek-plugin/dist/index.js" + # Install the packed plugin into this disposable profile first; loading + # raw dist bypasses Harness's peer dependency installation. + name: "/tmp/mem0-dsh-dev/profiles/headless/node_modules/@mem0/deepseek-plugin/dist/index.js" config: # apiKey is read from the MEM0_API_KEY env var when omitted here. userId: "your-user-id" + # autoRecall: true # optional; enabled by default + # autoCapture: true # optional; enabled by default # host: "https://your-onprem.mem0.ai" # optional: Platform on-prem / dedicated base URL diff --git a/integrations/deepseek-plugin/package.json b/integrations/deepseek-plugin/package.json index b64e4a4e7..ea00bb9c5 100644 --- a/integrations/deepseek-plugin/package.json +++ b/integrations/deepseek-plugin/package.json @@ -1,6 +1,6 @@ { "name": "@mem0/deepseek-plugin", - "version": "0.1.1", + "version": "0.3.0", "description": "Mem0 long-term memory as a native DeepSeek Harness (Cordis) plugin.", "type": "module", "license": "Apache-2.0", @@ -31,7 +31,6 @@ }, "files": [ "dist", - "src", "README.md", "LICENSE" ], @@ -46,11 +45,17 @@ }, "peerDependencies": { "@deepseek-ai/cordis": "*", + "@deepseek-ai/dsh-agent": "*", + "@deepseek-ai/dsh-session": "*", + "@deepseek-ai/dsh-system-prompt": "*", "@deepseek-ai/dsh-tools": "*" }, "devDependencies": { "@deepseek-ai/cordis": "4.0.1", - "@deepseek-ai/dsh-tools": "0.0.1-rc.1", + "@deepseek-ai/dsh-agent": "0.1.1-rc.2", + "@deepseek-ai/dsh-session": "0.1.1-rc.2", + "@deepseek-ai/dsh-system-prompt": "0.1.1-rc.2", + "@deepseek-ai/dsh-tools": "0.1.1-rc.2", "@types/node": "^22.15.0", "tsup": "^8.5.0", "typescript": "^5.6.0", diff --git a/integrations/deepseek-plugin/pnpm-lock.yaml b/integrations/deepseek-plugin/pnpm-lock.yaml index 26bc065c8..1444b53d4 100644 --- a/integrations/deepseek-plugin/pnpm-lock.yaml +++ b/integrations/deepseek-plugin/pnpm-lock.yaml @@ -15,9 +15,18 @@ importers: '@deepseek-ai/cordis': specifier: 4.0.1 version: 4.0.1 + '@deepseek-ai/dsh-agent': + specifier: 0.1.1-rc.2 + version: 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)(@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))(@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)) + '@deepseek-ai/dsh-session': + specifier: 0.1.1-rc.2 + version: 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1) + '@deepseek-ai/dsh-system-prompt': + specifier: 0.1.1-rc.2 + version: 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1) '@deepseek-ai/dsh-tools': - specifier: 0.0.1-rc.1 - version: 0.0.1-rc.1(@deepseek-ai/cordis@4.0.1) + specifier: 0.1.1-rc.2 + version: 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)(@deepseek-ai/dsh-agent@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)(@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))(@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)))(@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))(@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)) '@types/node': specifier: ^22.15.0 version: 22.20.1 @@ -48,18 +57,47 @@ packages: '@deepseek-ai/cosmokit@1.8.2': resolution: {integrity: sha512-muBOKtSrUKU5m/xpq8ZXWL6hQ/jgd4PhU2PqH97bcxIiLEJfNwZOGQEx4t/aS/GgxRAR+ra9pMHPMtTHU4sqqA==} - '@deepseek-ai/dsh-tools@0.0.1-rc.1': - resolution: {integrity: sha512-mrtq+UXc6UVidrGo4GM8RAH0Gqc4iFOmtPeQAz59Phhjl3EmzkFI0Mo6dQlHQamrc2J9jppPtGtUzW9qHmwxDQ==} + '@deepseek-ai/dsh-agent@0.1.1-rc.2': + resolution: {integrity: sha512-cC7lnJe7JgPFcreNXxcxLMxQd78LnpVO9ZXROjZsGRQN1zGH6i/DduI892F1am85IfzzO+XTxMwwUHmfwamb0g==} peerDependencies: - '@deepseek-ai/cordis': ^4.0.1-rc.1 - '@deepseek-ai/dsh-agent': ^0.0.1-rc.1 - '@deepseek-ai/dsh-code-runtime': ^0.0.1-rc.1 - '@deepseek-ai/dsh-invariants': ^0.0.1-rc.1 - '@deepseek-ai/dsh-llm': ^0.0.1-rc.1 - '@deepseek-ai/dsh-scope': ^0.0.1-rc.1 - '@deepseek-ai/dsh-session': ^0.0.1-rc.1 - '@deepseek-ai/dsh-system-prompt': ^0.0.1-rc.1 - '@deepseek-ai/dsh-user-approval': ^0.0.1-rc.1 + '@deepseek-ai/cordis': ^4.0.1 + '@deepseek-ai/dsh-invariants': ^0.1.1-rc.2 + '@deepseek-ai/dsh-llm': ^0.1.1-rc.2 + '@deepseek-ai/dsh-scope': ^0.1.1-rc.2 + '@deepseek-ai/dsh-session': ^0.1.1-rc.2 + '@deepseek-ai/dsh-system-prompt': ^0.1.1-rc.2 + '@deepseek-ai/dsh-typert-protocol': ^0.1.1-rc.2 + + '@deepseek-ai/dsh-session@0.1.1-rc.2': + resolution: {integrity: sha512-4/cv6X9HPhm47eyRhCu/WZwzrtJKegk5J+0xaxcZ9i8S0smdxP57tqy8a0jkSshLQn7BzMFxneQrlYExrLrDhQ==} + peerDependencies: + '@deepseek-ai/cordis': ^4.0.1 + '@deepseek-ai/dsh-brand': ^0.1.1-rc.2 + '@deepseek-ai/dsh-invariants': ^0.1.1-rc.2 + '@deepseek-ai/dsh-llm': ^0.1.1-rc.2 + '@deepseek-ai/dsh-scope': ^0.1.1-rc.2 + '@deepseek-ai/dsh-typert-protocol': ^0.1.1-rc.2 + + '@deepseek-ai/dsh-system-prompt@0.1.1-rc.2': + resolution: {integrity: sha512-on4hjAlYI5uX9q7Sf95YkMMBVe6heywtA/H50ksrIMUub8U2B98hO9iQpHhjwIO1F1vu+5pLcPvRr6yUGGmtXQ==} + peerDependencies: + '@deepseek-ai/cordis': ^4.0.1 + '@deepseek-ai/dsh-invariants': ^0.1.1-rc.2 + '@deepseek-ai/dsh-llm': ^0.1.1-rc.2 + '@deepseek-ai/dsh-scope': ^0.1.1-rc.2 + + '@deepseek-ai/dsh-tools@0.1.1-rc.2': + resolution: {integrity: sha512-0GGL4D55MwYDepzZMOI3L0ycu5b2qr96GL0Y7snwhAnpK2Di61rbX3fJE+PB3ZrovGX0csIRdt9n3iJZDVtDrw==} + peerDependencies: + '@deepseek-ai/cordis': ^4.0.1 + '@deepseek-ai/dsh-agent': ^0.1.1-rc.2 + '@deepseek-ai/dsh-code-runtime': ^0.1.1-rc.2 + '@deepseek-ai/dsh-invariants': ^0.1.1-rc.2 + '@deepseek-ai/dsh-llm': ^0.1.1-rc.2 + '@deepseek-ai/dsh-scope': ^0.1.1-rc.2 + '@deepseek-ai/dsh-session': ^0.1.1-rc.2 + '@deepseek-ai/dsh-system-prompt': ^0.1.1-rc.2 + '@deepseek-ai/dsh-user-approval': ^0.1.1-rc.2 '@deepseek-ai/schemastery@3.18.1': resolution: {integrity: sha512-Qn0FCSwCQnpnj6SB31I6i2sIKgKWnkbJM8O0EU91Gv2UsYVvtZTl6IA0sCwk2e2MZf5S8w5hpq9QkeVvK9qwxg==} @@ -1238,9 +1276,27 @@ snapshots: '@deepseek-ai/cosmokit@1.8.2': {} - '@deepseek-ai/dsh-tools@0.0.1-rc.1(@deepseek-ai/cordis@4.0.1)': + '@deepseek-ai/dsh-agent@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)(@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))(@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))': dependencies: '@deepseek-ai/cordis': 4.0.1 + '@deepseek-ai/dsh-session': 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1) + '@deepseek-ai/dsh-system-prompt': 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1) + + '@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)': + dependencies: + '@deepseek-ai/cordis': 4.0.1 + + '@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)': + dependencies: + '@deepseek-ai/cordis': 4.0.1 + '@deepseek-ai/schemastery': 3.18.1 + + '@deepseek-ai/dsh-tools@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)(@deepseek-ai/dsh-agent@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)(@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))(@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)))(@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))(@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))': + dependencies: + '@deepseek-ai/cordis': 4.0.1 + '@deepseek-ai/dsh-agent': 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)(@deepseek-ai/dsh-session@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1))(@deepseek-ai/dsh-system-prompt@0.1.1-rc.2(@deepseek-ai/cordis@4.0.1)) + '@deepseek-ai/dsh-session': 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1) + '@deepseek-ai/dsh-system-prompt': 0.1.1-rc.2(@deepseek-ai/cordis@4.0.1) '@deepseek-ai/schemastery': 3.18.1 '@deepseek-ai/schemastery@3.18.1': diff --git a/integrations/deepseek-plugin/src/formatting.ts b/integrations/deepseek-plugin/src/formatting.ts index 63296a185..79bb6283f 100644 --- a/integrations/deepseek-plugin/src/formatting.ts +++ b/integrations/deepseek-plugin/src/formatting.ts @@ -1,69 +1,7 @@ -/** - * Compact, token-cheap rendering of Mem0 results for the model context. - * - * Mirrors the format used by the sibling Mem0 plugins - * (integrations/pi-agent-plugin/src/memory/formatting.ts) so a memory reads the - * same way across every harness: `[category] text (age) [mem0:id]`. Dumping the - * raw search envelope instead would spend most of the tokens on JSON scaffolding. - */ - -export interface MemoryLike { - id: string; - memory?: string; - categories?: string[]; - createdAt?: Date | string; -} - -export function formatAge(date: Date | string): string { - const d = typeof date === "string" ? new Date(date) : date; - const ms = Date.now() - d.getTime(); - const minutes = Math.floor(ms / 60_000); - if (minutes < 60) return `${minutes}m ago`; - const hours = Math.floor(minutes / 60); - if (hours < 24) return `${hours}h ago`; - const days = Math.floor(hours / 24); - return `${days}d ago`; -} - -export function formatMemoryCompact(mem: MemoryLike): string { - const cat = mem.categories?.[0] ?? "uncategorized"; - const age = mem.createdAt ? ` (${formatAge(mem.createdAt)})` : ""; - return `[${cat}] ${mem.memory ?? "(empty)"}${age} [mem0:${mem.id}]`; -} - -export function formatMemoryList(memories: MemoryLike[]): string { - if (memories.length === 0) return "No memories found."; - return memories - .map((m, i) => `${i + 1}. ${formatMemoryCompact(m)}`) - .join("\n"); -} - -/** - * One-line confirmation for a write. - * - * `client.add` hits the async `/v3/memories/add/` endpoint, which returns - * `{ event_id, status: "PENDING" }` — extraction runs server-side *after* the - * call returns, so the extracted memories are not in this response. Report the - * write as queued in that case; only render a list when the backend actually - * returns memories (older / OSS shapes). - */ -export function formatAddResult(result: unknown): string { - const items: MemoryLike[] = Array.isArray(result) - ? (result as MemoryLike[]) - : ((result as { results?: MemoryLike[] } | null)?.results ?? - (result ? [result as MemoryLike] : [])); - - const pending = items.find( - (r) => (r as { status?: string }).status === "PENDING", - ) as { eventId?: string; event_id?: string } | undefined; - if (pending) { - // The SDK camel-cases response keys (event_id -> eventId); accept either. - const id = pending.eventId ?? pending.event_id; - const evt = id ? ` (event ${id})` : ""; - return `Memory queued for background extraction${evt}; it will be searchable shortly.`; - } - - if (items.length === 0) return "Memory stored."; - const noun = items.length === 1 ? "memory" : "memories"; - return `Stored ${items.length} ${noun}:\n${formatMemoryList(items)}`; -} +export { + formatAddResult, + formatAge, + formatMemoryCompact, + formatMemoryList, +} from "../../agent-plugin-core/typescript/src/formatting.ts"; +export type { MemoryLike } from "../../agent-plugin-core/typescript/src/formatting.ts"; diff --git a/integrations/deepseek-plugin/src/index.ts b/integrations/deepseek-plugin/src/index.ts index 6cb9bef12..bea70c02c 100644 --- a/integrations/deepseek-plugin/src/index.ts +++ b/integrations/deepseek-plugin/src/index.ts @@ -6,20 +6,24 @@ * - `add_memory` stores a fact for future sessions * * A plugin is a Cordis module that exports `apply(ctx, config)`. Declaring - * `inject = ['tools']` holds the plugin until the harness tool registry exists; + * `inject = ['tools', 'systemPrompt']` holds the plugin until the harness services exist; * tools registered via `ctx.tools.register(...)` are auto-unregistered when the * plugin unmounts (Cordis revertible effects). */ import type { Context } from "@deepseek-ai/cordis"; +import type {} from "@deepseek-ai/dsh-agent"; +import type { PromptAssembly } from "@deepseek-ai/dsh-system-prompt"; +import type {} from "@deepseek-ai/dsh-session"; import { defineTool } from "@deepseek-ai/dsh-tools"; import { MemoryClient } from "mem0ai"; import { formatMemoryList, formatAddResult } from "./formatting.ts"; import { truncateOutput } from "./output.ts"; import { resolveSearchFilters, resolveAddParams } from "./scoping.ts"; import { captureEvent, errorKind } from "./telemetry.ts"; +import { createMemoryLifecycle } from "../../agent-plugin-core/typescript/src/lifecycle.ts"; export const name = "mem0"; -export const inject = ["tools"]; +export const inject = ["tools", "systemPrompt"]; // Tags writes so Mem0's backend attributes them to this integration in // telemetry. The backend keeps recognized values via its KNOWN_EVENT_SOURCES @@ -29,14 +33,32 @@ export const inject = ["tools"]; const SOURCE = "DEEPSEEK_HARNESS"; const DEFAULT_SEARCH_LIMIT = 10; +const AUTO_RECALL_LIMIT = 5; + +interface HarnessMessage { + role: string; + content?: unknown; + source?: { kind?: string }; +} + +interface SessionState { + lifecycle: ReturnType; + messages: HarnessMessage[]; +} export interface Config { /** Mem0 API key. Defaults to the MEM0_API_KEY env var. */ apiKey?: string; /** Default entity that owns the memories (Mem0 user scope). */ userId: string; + /** Explicitly allow model-selected cross-user access. Defaults to false. */ + allowUserOverride?: boolean; /** Optional Mem0 Platform base-URL override (on-prem / dedicated); defaults to api.mem0.ai. Not a switch to self-hosted OSS. */ host?: string; + /** Recall relevant memory before each model request. Defaults to true. */ + autoRecall?: boolean; + /** Store completed turns automatically. Defaults to true. */ + autoCapture?: boolean; } // Both tools return a single text string; the render is identical, so lift it @@ -55,7 +77,7 @@ const scopeParams = { userId: { type: "string", description: - "Entity that owns the memory. Defaults to the plugin's configured userId; set this only to read or write another user's memories.", + "Entity that owns the memory. Defaults to the plugin's configured userId; cross-user overrides require allowUserOverride in plugin configuration.", }, agentId: { type: "string", @@ -72,8 +94,8 @@ export function apply(ctx: Context, config: Config): void { if (!apiKey) { throw new Error("deepseek-plugin: set config.apiKey or the MEM0_API_KEY env var"); } - const userId = config.userId; - if (!userId) { + const userId = config.userId?.trim(); + if (!userId || /^\*+$/.test(userId)) { throw new Error("deepseek-plugin: config.userId is required"); } @@ -81,8 +103,95 @@ export function apply(ctx: Context, config: Config): void { apiKey, ...(config.host ? { host: config.host } : {}), }); + const toolLifecycle = createMemoryLifecycle(); + const sessionStates = new WeakMap(); + const stateFor = (session: object): SessionState => { + let state = sessionStates.get(session); + if (!state) { + const lifecycle = createMemoryLifecycle(); + lifecycle.beginSession(); + state = { lifecycle, messages: [] }; + sessionStates.set(session, state); + } + return state; + }; - captureEvent("deepseek.plugin.mounted", { has_host: Boolean(config.host) }, client); + captureEvent("deepseek.plugin.mounted", { + has_host: Boolean(config.host), + auto_recall: config.autoRecall !== false, + auto_capture: config.autoCapture !== false, + }, client); + + if (config.autoRecall !== false) { + ctx.on("system-prompt/assemble", async (_input, context, next): Promise => { + const assembly = await next(); + const { agent, signal } = context; + if (!agent || signal?.aborted) return assembly; + + const state = stateFor(agent.session); + const prompt = state.lifecycle.prepareConversation( + agent.session + .deriveMessages() + .filter((message) => message.role === "user" && message.source?.kind === "user"), + ).at(-1)?.content; + if (!prompt) return assembly; + + const memoryContext = await state.lifecycle.recall(prompt, true, async (query) => { + const started = Date.now(); + try { + const result = await client.search(query, { + filters: resolveSearchFilters({}, userId), + topK: AUTO_RECALL_LIMIT, + }); + captureEvent("deepseek.recall.auto", { + success: true, + duration_ms: Date.now() - started, + result_count: result.results?.length ?? 0, + }, client); + return result; + } catch (err) { + captureEvent("deepseek.recall.auto", { + success: false, + duration_ms: Date.now() - started, + error_kind: errorKind(err), + }, client); + throw err; + } + }); + if (!memoryContext || signal?.aborted) return assembly; + return { + ...assembly, + contexts: [...assembly.contexts, { name: "mem0:recall", text: memoryContext }], + }; + }); + } + + if (config.autoCapture !== false) { + ctx.on("session/event", (session, event) => { + const state = stateFor(session); + if (event.type === "turn/start") { + state.messages = []; + } else if (event.type === "user/message" && event.data.source.kind === "user") { + state.messages.push(event.data); + } else if (event.type === "assistant/message") { + state.messages.push(event.data.message); + } else if (event.type === "turn/end") { + const conversation = state.lifecycle.prepareConversation(state.messages); + state.messages = []; + if (event.data.reason.kind !== "completed" || conversation.length === 0) return; + void client + .add(conversation, { userId, source: SOURCE }) + .then(() => captureEvent("deepseek.capture.auto", { + success: true, + message_count: conversation.length, + }, client)) + .catch((err: unknown) => captureEvent("deepseek.capture.auto", { + success: false, + error_kind: errorKind(err), + }, client)); + } + }); + } // Recall. The platform rejects top-level entity params on search, so scope // goes inside `filters` (unlike add below, which takes them top-level). @@ -101,11 +210,15 @@ export function apply(ctx: Context, config: Config): void { }, output: textOutput, async execute({ query, limit, userId: u, agentId, runId }) { + if (u?.trim() && u.trim() !== userId && config.allowUserOverride !== true) { + throw new Error("Cross-user access requires allowUserOverride in plugin configuration."); + } + const safeQuery = toolLifecycle.prepareUserText(query); const filters = resolveSearchFilters({ userId: u, agentId, runId }, userId); const topK = limit && limit > 0 ? limit : DEFAULT_SEARCH_LIMIT; const started = Date.now(); try { - const { results } = await client.search(query, { filters, topK }); + const { results } = await client.search(safeQuery, { filters, topK }); captureEvent( "deepseek.tool.search_memory", { @@ -150,13 +263,19 @@ export function apply(ctx: Context, config: Config): void { }, output: textOutput, async execute({ text, userId: u, agentId, runId }) { + if (u?.trim() && u.trim() !== userId && config.allowUserOverride !== true) { + throw new Error("Cross-user access requires allowUserOverride in plugin configuration."); + } const addParams = resolveAddParams({ userId: u, agentId, runId }, userId); const started = Date.now(); try { - const result = await client.add([{ role: "user", content: text }], { - ...addParams, - source: SOURCE, - }); + const result = await client.add( + [{ role: "user", content: toolLifecycle.prepareUserText(text) }], + { + ...addParams, + source: SOURCE, + }, + ); captureEvent( "deepseek.tool.add_memory", { diff --git a/integrations/deepseek-plugin/src/output.ts b/integrations/deepseek-plugin/src/output.ts index 44eee7c30..7ec64eff9 100644 --- a/integrations/deepseek-plugin/src/output.ts +++ b/integrations/deepseek-plugin/src/output.ts @@ -1,33 +1,5 @@ -/** - * Hard cap on tool output before it reaches the model context. - * - * Same guard the sibling plugins apply (200 lines / 50KB, see - * integrations/pi-agent-plugin/src/memory/tools.ts): a large recall or a wide - * result set can otherwise flood the context window in a single tool call. - */ - -export const MAX_OUTPUT_LINES = 200; -export const MAX_OUTPUT_BYTES = 50_000; - -export function truncateOutput(text: string): string { - const lines = text.split("\n"); - if (lines.length <= MAX_OUTPUT_LINES && text.length <= MAX_OUTPUT_BYTES) { - return text; - } - - const kept = lines.slice(0, MAX_OUTPUT_LINES); - let result = kept.join("\n"); - const byteCapped = result.length > MAX_OUTPUT_BYTES; - if (byteCapped) { - result = result.slice(0, MAX_OUTPUT_BYTES); - } - - const dropped = lines.length - kept.length; - const reasons: string[] = []; - if (dropped > 0) reasons.push(`showing ${kept.length} of ${lines.length} lines`); - if (byteCapped) reasons.push(`cut at ${Math.floor(MAX_OUTPUT_BYTES / 1000)}KB`); - if (reasons.length > 0) { - result += `\n\n[Output truncated: ${reasons.join(", ")}]`; - } - return result; -} +export { + MAX_OUTPUT_BYTES, + MAX_OUTPUT_LINES, + truncateOutput, +} from "../../agent-plugin-core/typescript/src/formatting.ts"; diff --git a/integrations/deepseek-plugin/src/scoping.ts b/integrations/deepseek-plugin/src/scoping.ts index ee5bfd81d..ac471b0aa 100644 --- a/integrations/deepseek-plugin/src/scoping.ts +++ b/integrations/deepseek-plugin/src/scoping.ts @@ -1,53 +1,5 @@ -/** - * Per-call memory scoping. - * - * The plugin is mounted with one default `userId`, but a single harness install - * can serve more than one entity, so both tools accept optional `userId` / - * `agentId` / `runId` params that override the mount-time default per call. - * A missing or blank param falls back to the configured user. - * - * The two call sites need different key casing, and it is deliberate rather than - * incidental: search passes scope inside `filters`, sent to the platform raw, so - * it must be snake_case; add takes the entity params top-level, through the - * SDK's camel->snake converter, so it must be camelCase. Keeping the split - * explicit (like integrations/pi-agent-plugin/src/memory/scoping.ts) means the - * asymmetry is visible in the code, not load-bearing on a converter no-op. - */ - -export interface EntityParams { - userId?: string; - agentId?: string; - runId?: string; -} - -const clean = (v: string | undefined) => v?.trim() || undefined; - -/** Search: snake_case, spread into `filters` and passed to the platform raw. */ -export function resolveSearchFilters( - params: EntityParams, - defaultUserId: string, -): Record { - const filters: Record = { - user_id: clean(params.userId) ?? defaultUserId, - }; - const agentId = clean(params.agentId); - if (agentId) filters.agent_id = agentId; - const runId = clean(params.runId); - if (runId) filters.run_id = runId; - return filters; -} - -/** Add: camelCase, top-level params run through the SDK's camel->snake converter. */ -export function resolveAddParams( - params: EntityParams, - defaultUserId: string, -): Record { - const out: Record = { - userId: clean(params.userId) ?? defaultUserId, - }; - const agentId = clean(params.agentId); - if (agentId) out.agentId = agentId; - const runId = clean(params.runId); - if (runId) out.runId = runId; - return out; -} +export { + entityAddParams as resolveAddParams, + entitySearchFilters as resolveSearchFilters, +} from "../../agent-plugin-core/typescript/src/identity.ts"; +export type { EntityParams } from "../../agent-plugin-core/typescript/src/identity.ts"; diff --git a/integrations/deepseek-plugin/src/telemetry.ts b/integrations/deepseek-plugin/src/telemetry.ts index b963e62ce..24cd868a3 100644 --- a/integrations/deepseek-plugin/src/telemetry.ts +++ b/integrations/deepseek-plugin/src/telemetry.ts @@ -1,173 +1,98 @@ -/** - * Anonymous usage telemetry for the DeepSeek Harness plugin. - * - * Fire-and-forget PostHog events over native fetch, batched and flushed every - * 5 seconds, at 10 queued events, or on process exit. Never throws, never logs. - * - * Events carry only tool names, durations, counts, and coarse failure kinds: - * never queries, memory text, filters, or API keys. - * - * Disable with MEM0_TELEMETRY=false. - */ import { randomUUID } from "node:crypto"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"; -const POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/"; -const FLUSH_INTERVAL_MS = 5_000; -const FLUSH_THRESHOLD = 10; -const SEND_TIMEOUT_MS = 3_000; +import { + createTelemetry, + errorKind, + isTelemetryEnabled, +} from "../../agent-plugin-core/typescript/src/telemetry.ts"; -const PLUGIN_VERSION = ((): string => { +export { errorKind, isTelemetryEnabled }; + +const PLUGIN_VERSION = (() => { try { - return JSON.parse( - fs.readFileSync(new URL("../package.json", import.meta.url), "utf-8"), - ).version; + return JSON.parse(fs.readFileSync(new URL("../package.json", import.meta.url), "utf-8")).version; } catch { return "unknown"; } })(); -/** The Mem0 SDK resolves this to the account email once ping() lands, so events join other Mem0 surfaces. */ export interface TelemetryIdentity { telemetryId?: string; } -let queue: Record[] = []; -let flushTimer: ReturnType | undefined; -let exitHandlerInstalled = false; let cachedAnonymousId: string | undefined; let identified = false; +let currentDistinctId = ""; function identityPath(): string { return path.join(os.homedir(), ".mem0", "deepseek-plugin-telemetry.json"); } -export function isTelemetryEnabled(): boolean { - const value = process.env.MEM0_TELEMETRY?.toLowerCase(); - return value !== "false" && value !== "0" && value !== "no" && value !== "off"; -} - function anonymousId(): string { if (cachedAnonymousId) return cachedAnonymousId; try { const stored = JSON.parse(fs.readFileSync(identityPath(), "utf-8")); if (typeof stored.anonymousId === "string" && stored.anonymousId) { - cachedAnonymousId = stored.anonymousId; - return cachedAnonymousId!; + return (cachedAnonymousId = stored.anonymousId); } } catch { - /* first run, or an unreadable identity file */ + // First run or unreadable identity file. } const created = `deepseek-anon-${randomUUID().replace(/-/g, "")}`; try { - const target = identityPath(); - fs.mkdirSync(path.dirname(target), { recursive: true }); - fs.writeFileSync(target, JSON.stringify({ anonymousId: created }), "utf-8"); + fs.mkdirSync(path.dirname(identityPath()), { recursive: true }); + fs.writeFileSync(identityPath(), JSON.stringify({ anonymousId: created }), "utf-8"); } catch { - /* an unwritable home directory must not break the plugin */ + // An unwritable home directory must not break the plugin. } - cachedAnonymousId = created; - return created; + return (cachedAnonymousId = created); } -/** Merge the pre-ping anonymous history into the account once the email resolves. */ -function identifyEvent(distinctId: string): Record | undefined { +function previousAnonymousId(distinctId: string): string | undefined { if (identified || distinctId.startsWith("deepseek-anon-")) return undefined; identified = true; - let storedAnonymousId: string | undefined; try { - storedAnonymousId = JSON.parse(fs.readFileSync(identityPath(), "utf-8")).anonymousId; + const stored = JSON.parse(fs.readFileSync(identityPath(), "utf-8")).anonymousId; fs.unlinkSync(identityPath()); + cachedAnonymousId = undefined; + return typeof stored === "string" && stored ? stored : undefined; } catch { return undefined; } - cachedAnonymousId = undefined; - if (!storedAnonymousId) return undefined; - return { - event: "$identify", - distinct_id: distinctId, - properties: { $anon_distinct_id: storedAnonymousId, $lib: "posthog-node" }, - }; } -export function flushEvents(): void { - if (queue.length === 0) return; - const batch = queue; - queue = []; - fetch(POSTHOG_BATCH_URL, { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ api_key: POSTHOG_API_KEY, batch }), - signal: AbortSignal.timeout(SEND_TIMEOUT_MS), - }).catch(() => { - /* telemetry never surfaces its own failures */ - }); -} - -function scheduleFlush(): void { - if (!flushTimer) { - flushTimer = setInterval(flushEvents, FLUSH_INTERVAL_MS); - flushTimer.unref?.(); - } - if (!exitHandlerInstalled) { - exitHandlerInstalled = true; - process.on("beforeExit", flushEvents); - } -} - -export function errorKind(error: unknown): string { - const text = (error instanceof Error ? error.message : String(error)).toLowerCase(); - if (text.includes("timeout") || text.includes("aborted")) return "timeout"; - if (text.includes("401") || text.includes("403") || text.includes("unauthor")) return "auth"; - if (text.includes("429") || text.includes("rate limit")) return "rate-limited"; - if (/50[0234]/.test(text)) return "server-error"; - if (text.includes("400") || text.includes("422")) return "bad-request"; - if (text.includes("fetch failed") || text.includes("enotfound")) return "network"; - return error instanceof Error ? error.constructor.name : "other"; -} +const telemetry = createTelemetry({ + host: "deepseek", + source: "DEEPSEEK_HARNESS", + version: PLUGIN_VERSION, + distinctId: () => currentDistinctId, +}); export function captureEvent( event: string, properties: Record, client: TelemetryIdentity, ): void { - if (!isTelemetryEnabled()) return; - try { - const distinctId = client.telemetryId || anonymousId(); - const identify = identifyEvent(distinctId); - if (identify) queue.push(identify); - queue.push({ - event, - distinct_id: distinctId, - properties: { - source: "DEEPSEEK_HARNESS", - language: "node", - plugin_version: PLUGIN_VERSION, - node_version: process.version, - os: process.platform, - $process_person_profile: false, - $lib: "posthog-node", - ...properties, - }, - }); - scheduleFlush(); - if (queue.length >= FLUSH_THRESHOLD) flushEvents(); - } catch { - /* telemetry never breaks a tool call */ - } + currentDistinctId = client.telemetryId || anonymousId(); + const anonymous = previousAnonymousId(currentDistinctId); + if (anonymous) telemetry.capture("$identify", { $anon_distinct_id: anonymous }); + telemetry.capture(event, properties); +} + +export function flushEvents(): void { + void telemetry.flush(); } export function _queueForTesting(): Record[] { - return queue; + return telemetry.queueForTesting(); } export function _resetForTesting(): void { - queue = []; - clearInterval(flushTimer); - flushTimer = undefined; + telemetry.resetForTesting(); cachedAnonymousId = undefined; + currentDistinctId = ""; identified = false; } diff --git a/integrations/deepseek-plugin/tests/apply.test.ts b/integrations/deepseek-plugin/tests/apply.test.ts index 5c6c413be..797b69695 100644 --- a/integrations/deepseek-plugin/tests/apply.test.ts +++ b/integrations/deepseek-plugin/tests/apply.test.ts @@ -25,15 +25,28 @@ interface RegisteredTool { execute(args: unknown, exec: unknown): Promise; } +type HarnessListener = (...args: any[]) => unknown; + function applyAndCollect(config: Config): Map { const tools = new Map(); const ctx = { tools: { register: (t: RegisteredTool) => tools.set(t.name, t) }, + on: vi.fn(), }; apply(ctx as never, config); return tools; } +function applyAndCollectListeners(config: Config): Map { + const listeners = new Map(); + const ctx = { + tools: { register: vi.fn() }, + on: (event: string, listener: HarnessListener) => listeners.set(event, listener), + }; + apply(ctx as never, config); + return listeners; +} + let savedKey: string | undefined; let savedTelemetry: string | undefined; @@ -68,6 +81,97 @@ describe("apply() config validation", () => { }); }); +describe("Harness lifecycle", () => { + it("automatically recalls memory into the prompt for the latest human message", async () => { + mockSearch.mockResolvedValue({ + results: [{ id: "m1", memory: "Likes tea" }], + }); + const listeners = applyAndCollectListeners({ apiKey: "k", userId: "u" }); + const assemble = listeners.get("system-prompt/assemble")!; + const base = { sections: [], contexts: [], tools: [], variables: {} }; + + const result = await assemble( + base, + { + agent: { + session: { + deriveMessages: () => [ + { + role: "user", + content: [{ type: "text", text: "What do I drink?" }], + source: { kind: "user" }, + }, + ], + }, + }, + }, + async () => base, + ); + + expect(mockSearch).toHaveBeenCalledWith("What do I drink?", { + filters: { user_id: "u" }, + topK: 5, + }); + expect(result).toMatchObject({ + contexts: [{ name: "mem0:recall", text: expect.stringContaining("Likes tea") }], + }); + }); + + it("automatically captures a completed human and assistant turn", async () => { + mockAdd.mockResolvedValue({ eventId: "evt-1", status: "PENDING" }); + const listeners = applyAndCollectListeners({ apiKey: "k", userId: "u" }); + const onSessionEvent = listeners.get("session/event")!; + const session = {}; + + onSessionEvent(session, { type: "turn/start", data: { turn: 1 } }); + onSessionEvent(session, { + type: "user/message", + data: { + role: "user", + content: [{ type: "text", text: "Remember I like tea" }], + source: { kind: "user" }, + }, + }); + onSessionEvent(session, { + type: "assistant/message", + data: { + turn: 1, + message: { + role: "assistant", + content: [{ type: "text", text: "I will remember that." }], + source: { kind: "model" }, + }, + }, + }); + onSessionEvent(session, { + type: "turn/end", + data: { turn: 1, reason: { kind: "completed" } }, + }); + + await vi.waitFor(() => { + expect(mockAdd).toHaveBeenCalledWith( + [ + { role: "user", content: "Remember I like tea" }, + { role: "assistant", content: "I will remember that." }, + ], + { userId: "u", source: "DEEPSEEK_HARNESS" }, + ); + }); + }); + + it("can disable automatic recall and capture without removing the memory tools", () => { + const listeners = applyAndCollectListeners({ + apiKey: "k", + userId: "u", + autoRecall: false, + autoCapture: false, + }); + + expect(listeners.has("system-prompt/assemble")).toBe(false); + expect(listeners.has("session/event")).toBe(false); + }); +}); + describe("search_memory tool", () => { it("returns a formatted list scoped to the configured user", async () => { mockSearch.mockResolvedValue({ @@ -87,7 +191,7 @@ describe("search_memory tool", () => { it("honors a per-call userId override and limit", async () => { mockSearch.mockResolvedValue({ results: [] }); - const tools = applyAndCollect({ apiKey: "k", userId: "u" }); + const tools = applyAndCollect({ apiKey: "k", userId: "u", allowUserOverride: true }); await tools.get("search_memory")!.execute({ query: "x", userId: "alice", limit: 3 }, {}); @@ -136,6 +240,18 @@ describe("add_memory tool", () => { expect(out).toContain("[mem0:m1]"); }); + it("redacts credentials before storing explicit memory", async () => { + mockAdd.mockResolvedValue([]); + const tools = applyAndCollect({ apiKey: "k", userId: "u" }); + + await tools.get("add_memory")!.execute({ text: "api_key=do-not-store-this" }, {}); + + expect(mockAdd).toHaveBeenCalledWith( + [{ role: "user", content: "api_key=[REDACTED]" }], + { userId: "u", source: "DEEPSEEK_HARNESS" }, + ); + }); + it("returns a graceful failure line on error", async () => { mockAdd.mockRejectedValue(new Error("boom")); const tools = applyAndCollect({ apiKey: "k", userId: "u" }); @@ -146,3 +262,13 @@ describe("add_memory tool", () => { expect(out).toContain("boom"); }); }); + +describe("tool user ownership", () => { + it("rejects cross-user reads and writes before contacting Mem0", async () => { + const tools = applyAndCollect({ apiKey: "k", userId: "u" }); + await expect(tools.get("search_memory")!.execute({ query: "x", userId: "other" }, {})).rejects.toThrow(/allowUserOverride/); + await expect(tools.get("add_memory")!.execute({ text: "x", userId: "other" }, {})).rejects.toThrow(/allowUserOverride/); + expect(mockSearch).not.toHaveBeenCalled(); + expect(mockAdd).not.toHaveBeenCalled(); + }); +}); diff --git a/integrations/deepseek-plugin/tests/telemetry.test.ts b/integrations/deepseek-plugin/tests/telemetry.test.ts index d81b2869e..14d77d6eb 100644 --- a/integrations/deepseek-plugin/tests/telemetry.test.ts +++ b/integrations/deepseek-plugin/tests/telemetry.test.ts @@ -31,7 +31,10 @@ interface RegisteredTool { function applyAndCollect(config: Config): Map { const tools = new Map(); - apply({ tools: { register: (t: RegisteredTool) => tools.set(t.name, t) } } as never, config); + apply({ + tools: { register: (t: RegisteredTool) => tools.set(t.name, t) }, + on: vi.fn(), + } as never, config); return tools; } @@ -190,7 +193,7 @@ describe("tool instrumentation", () => { it("records a write with its size but not its text", async () => { mockAdd.mockResolvedValue([{ id: "m1", memory: "Fact" }]); - const tools = applyAndCollect({ apiKey: "k", userId: "u" }); + const tools = applyAndCollect({ apiKey: "k", userId: "u", allowUserOverride: true }); await tools.get("add_memory")!.execute({ text: "secret fact", userId: "alice" }, {}); diff --git a/integrations/deepseek-plugin/tsconfig.json b/integrations/deepseek-plugin/tsconfig.json index 412c13bac..c36f17b65 100644 --- a/integrations/deepseek-plugin/tsconfig.json +++ b/integrations/deepseek-plugin/tsconfig.json @@ -6,7 +6,6 @@ "lib": ["ES2022"], "declaration": true, "outDir": "dist", - "rootDir": "src", "strict": true, "types": ["node"], "esModuleInterop": true, diff --git a/integrations/kimi-plugin/agents/sidekick.md b/integrations/kimi-plugin/agents/sidekick.md new file mode 100644 index 000000000..8032119b0 --- /dev/null +++ b/integrations/kimi-plugin/agents/sidekick.md @@ -0,0 +1,18 @@ +--- +name: sidekick +description: Coding subagent for focused implementation, investigation, testing, debugging, or review work. +whenToUse: Delegate a bounded engineering task that benefits from its own isolated context. +--- + +You are Mem0's coding sidekick. Complete only the bounded task the main agent +delegates to you and return a concise, self-contained result. + +Search Mem0 before work that may depend on prior repository decisions or user +preferences. Inspect the relevant repository rules and code, make changes when +asked, and run the smallest decisive validation. Do not claim Git worktree +isolation: Kimi provides a separate context, while filesystem isolation depends +on the caller's environment. + +Your final response must state the outcome, changed files, validation, and any +remaining risk. Do not commit, push, or open a pull request unless the caller +explicitly asks. diff --git a/integrations/kimi-plugin/core/flush_worker.py b/integrations/kimi-plugin/core/flush_worker.py new file mode 100644 index 000000000..6f7b9ecb2 --- /dev/null +++ b/integrations/kimi-plugin/core/flush_worker.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Detached remote checkpoint worker. + +Claude Code may cancel SessionEnd hooks as a print-mode process exits. The hook +therefore persists its input first and launches this process in a new session. +""" + +from __future__ import annotations + +import json +import os +import sys +import time +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + checkpoint_session, + configure_harness, + touch_handoff_heartbeat, +) + + +def main() -> int: + if len(sys.argv) != 2: + return 2 + handoff_path = Path(sys.argv[1]) + os.environ["MEM0_CODE_HANDOFF_PATH"] = str(handoff_path) + harness = os.environ.get("MEM0_PLUGIN_HARNESS") + if harness: + source_tag = os.environ.get("MEM0_PLUGIN_SOURCE_TAG", "") + configure_harness( + harness, + env_prefix=os.environ.get("MEM0_PLUGIN_ENV_PREFIX", ""), + data_dir_name=os.environ.get("MEM0_PLUGIN_DATA_DIR_NAME", ""), + source_tag=source_tag, + ) + telemetry.init(harness=harness, source_tag=source_tag.upper()) + completed = False + try: + payload = json.loads(handoff_path.read_text(encoding="utf-8")) + delay = float(payload.get("delay_seconds") or 0) + if delay > 0: + payload.pop("delay_seconds", None) + temporary = handoff_path.with_suffix(f".{os.getpid()}.tmp") + try: + temporary.write_text(json.dumps(payload), encoding="utf-8") + temporary.replace(handoff_path) + finally: + temporary.unlink(missing_ok=True) + time.sleep(delay) + if not handoff_path.exists(): + return 0 + hook_input = payload.get("hook_input") or {} + reason = str(payload.get("reason") or "checkpoint") + wait_for_inflight = bool(payload.get("wait_for_inflight")) + store = EvidenceStore() + try: + if wait_for_inflight: + session_id = str( + hook_input.get("session_id") or "unknown-session" + ) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + deadline = time.monotonic() + float( + os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120") + ) + while ( + store.has_inflight_flush(repo.identity, session_id) + and time.monotonic() < deadline + ): + touch_handoff_heartbeat() + time.sleep(0.25) + # Hooks capture the conversation before handoff; the worker only flushes it. + result = checkpoint_session(store, hook_input, reason) + print(json.dumps(result, sort_keys=True), flush=True) + completed = result.get("status") in { + "semantic-succeeded", + "explicitly-stored", + "nothing-to-flush", + } + finally: + store.close() + return 0 + finally: + telemetry.flush() + if completed: + try: + handoff_path.unlink() + except OSError: + pass + elif handoff_path.suffix == ".running": + try: + handoff_path.replace(handoff_path.with_suffix(".json")) + except OSError: + pass + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/kimi-plugin/core/hook_runner.py b/integrations/kimi-plugin/core/hook_runner.py new file mode 100644 index 000000000..2a2adb557 --- /dev/null +++ b/integrations/kimi-plugin/core/hook_runner.py @@ -0,0 +1,372 @@ +"""Shared hook orchestration for all Mem0 agent plugins.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import subprocess +import sys +import time +import uuid +from pathlib import Path + +import telemetry +from memory_core import ( + EvidenceStore, + _session_id, + api_key, + bounded, + cache_plugin_api_key, + checkpoint_session, + clear_stale_api_key_cache, + configure_harness, + data_dir, + detached_process_kwargs, + format_context, + harness_config, + record_session_start, + record_tool, + record_user_prompt, + redact, + search_memories, +) + +STALE_RUNNING_SECONDS = 300 +PENDING_EXPIRY_SECONDS = 7 * 24 * 60 * 60 +PENDING_LAUNCH_LIMIT = 5 +DEFAULT_IDLE_FLUSH_SECONDS = 300 + +_core_dir: Path = Path(__file__).resolve().parent + + +def read_hook_input() -> dict: + try: + value = json.load(sys.stdin) + return value if isinstance(value, dict) else {} + except (json.JSONDecodeError, OSError): + return {} + + +def default_record_stop(store: EvidenceStore, hook_input: dict): + """Record the assistant's response without transcript parsing.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + message = redact(hook_input.get("last_assistant_message", "")).strip() + if message: + store.record_assistant_response(repo, session_id, message) + return repo, session_id + + +def first_prompt_memory_output(store: EvidenceStore, hook_input: dict) -> dict: + """Search once before the agent handles the first prompt in a session.""" + repo, session_id, prompt, is_first_prompt = record_user_prompt(store, hook_input) + if not is_first_prompt: + return {} + try: + minimum_query_chars = int(os.environ.get("MEM0_CODE_MIN_QUERY_CHARS", "20")) + except ValueError: + minimum_query_chars = 20 + if len(prompt.strip()) < max(minimum_query_chars, 1): + return {} + result = search_memories( + store, repo, session_id, bounded(prompt, 6000), + top_k=5, operation="first-prompt-search", timeout=2, + ) + if not result.memories: + return {} + context = format_context( + result.memories, + "Mem0 found these relevant memories from earlier work in this repository:", + ) + telemetry.record( + "context_injected", + repo=repo, session_id=session_id, trigger="first-prompt", + memory_count=len(result.memories), context_chars=len(context), + prompt_chars=len(prompt), + ) + return { + "hookSpecificOutput": { + "hookEventName": "UserPromptSubmit", + "additionalContext": context, + }, + } + + +def _launch_handoff(handoff_path: Path) -> bool: + running_path = handoff_path.with_suffix(".running") + try: + handoff_path.replace(running_path) + except OSError: + return False + worker = _core_dir / "flush_worker.py" + log_path = data_dir() / "flush-worker.log" + log_handle = open(log_path, "a", encoding="utf-8") + harness = harness_config() + child_env = os.environ.copy() + child_env.update( + { + "MEM0_CODE_DATA_DIR": str(data_dir()), + "MEM0_PLUGIN_HARNESS": harness["name"], + "MEM0_PLUGIN_ENV_PREFIX": harness["env_prefix"], + "MEM0_PLUGIN_DATA_DIR_NAME": harness["data_dir_name"], + "MEM0_PLUGIN_SOURCE_TAG": harness["source_tag"], + } + ) + try: + subprocess.Popen( + [sys.executable, str(worker), str(running_path)], + stdin=subprocess.DEVNULL, + stdout=log_handle, stderr=log_handle, + close_fds=True, + env=child_env, + **detached_process_kwargs(), + ) + finally: + log_handle.close() + return True + + +def recover_pending_handoffs() -> int: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + now = time.time() + for running in pending_dir.glob("*.running"): + try: + if now - running.stat().st_mtime > STALE_RUNNING_SECONDS: + running.replace(running.with_suffix(".json")) + except OSError: + continue + recoverable = [] + for handoff in pending_dir.glob("*.json"): + try: + age = now - handoff.stat().st_mtime + except OSError: + continue + if age > PENDING_EXPIRY_SECONDS: + handoff.unlink(missing_ok=True) + continue + recoverable.append((age, handoff)) + recoverable.sort(key=lambda item: item[0], reverse=True) + launched = 0 + for _, handoff in recoverable[:PENDING_LAUNCH_LIMIT]: + launched += int(_launch_handoff(handoff)) + return launched + + +def refresh_pending_handoffs() -> None: + pending_dir = data_dir() / "pending" + if not pending_dir.is_dir(): + return + for pattern in ("*.json", "*.running"): + for handoff in pending_dir.glob(pattern): + try: + os.utime(handoff) + except OSError: + continue + + +def hand_off_flush( + hook_input: dict, reason: str, *, wait_for_inflight: bool = False, +) -> None: + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = ( + f"{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}\0{reason}" + ) + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + handoff_path = pending_dir / f"{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": reason, + "wait_for_inflight": wait_for_inflight, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + + +def automatic_flush_enabled() -> bool: + return os.environ.get("MEM0_CODE_AUTO_FLUSH", "true").lower() in { + "1", "true", "yes", "on", + } + + +def schedule_periodic_checkpoint( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + if ( + not automatic_flush_enabled() + or not api_key() + or not store.checkpoint_due(repo.identity, session_id) + ): + return False + if store.prepare_flush(repo, session_id, "periodic") is None: + return False + hand_off_flush(hook_input, "periodic") + return True + + +def _idle_flush_seconds() -> int: + try: + return max( + int(os.environ.get("MEM0_CODE_IDLE_FLUSH_SECONDS", str(DEFAULT_IDLE_FLUSH_SECONDS))), + 0, + ) + except ValueError: + return DEFAULT_IDLE_FLUSH_SECONDS + + +def schedule_idle_flush( + store: EvidenceStore, hook_input: dict, repo, session_id: str, +) -> bool: + delay = _idle_flush_seconds() + if delay <= 0 or not automatic_flush_enabled() or not api_key(): + return False + if store.has_inflight_flush(repo.identity, session_id): + return False + if not store.has_unflushed_events(repo.identity, session_id): + return False + pending_dir = data_dir() / "pending" + pending_dir.mkdir(parents=True, exist_ok=True) + material = f"idle\0{hook_input.get('cwd', '')}\0{hook_input.get('session_id', '')}" + digest = hashlib.sha256(material.encode()).hexdigest()[:24] + for old in pending_dir.glob(f"idle-{digest}*"): + old.unlink(missing_ok=True) + handoff_path = pending_dir / f"idle-{digest}-{uuid.uuid4().hex[:8]}.json" + temporary_path = handoff_path.with_suffix(".tmp") + temporary_path.write_text( + json.dumps({ + "hook_input": hook_input, + "reason": "idle", + "delay_seconds": delay, + }), + encoding="utf-8", + ) + temporary_path.replace(handoff_path) + _launch_handoff(handoff_path) + return True + + +def log_failure(exc: Exception) -> None: + try: + log_path = data_dir() / "plugin-errors.log" + with log_path.open("a", encoding="utf-8") as handle: + handle.write(f"{time.time():.3f} {type(exc).__name__}: {exc}\n") + except OSError: + pass + + +def run( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> int: + if record_stop_fn is None: + record_stop_fn = default_record_stop + if automatic_flush_reasons is None: + automatic_flush_reasons = {"session-end"} + + base_actions = ["session-start", "user-prompt", "post-tool", "stop", "flush"] + all_actions = base_actions + list((extra_actions or {}).keys()) + + parser = argparse.ArgumentParser() + parser.add_argument("action", choices=all_actions) + parser.add_argument("--reason", default="manual") + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + args = parser.parse_args() + + if args.harness: + configure_harness(args.harness) + telemetry.init(harness=args.harness) + + if args.plugin_data_dir: + os.environ[data_dir_env] = args.plugin_data_dir + + cache_plugin_api_key() + if args.action == "session-start": + clear_stale_api_key_cache() + + hook_input = read_hook_input() + store = EvidenceStore() + try: + if store.is_paused(): + if args.action == "session-start": + refresh_pending_handoffs() + telemetry.record("session_start", paused=True) + telemetry.spawn_flush() + return 0 + + if args.action == "session-start": + if telemetry.is_first_run(): + telemetry.record("install") + recovered = recover_pending_handoffs() + record_session_start(store, hook_input) + if recovered: + telemetry.record("handoff_recovered", count=recovered) + telemetry.spawn_flush() + elif args.action == "user-prompt": + output = first_prompt_memory_output(store, hook_input) + if output: + print(json.dumps(output)) + elif args.action == "post-tool": + record_tool(store, hook_input) + elif args.action == "stop": + repo, session_id = record_stop_fn(store, hook_input) + if not schedule_periodic_checkpoint(store, hook_input, repo, session_id): + schedule_idle_flush(store, hook_input, repo, session_id) + elif args.action == "flush": + automatic = args.reason in automatic_flush_reasons + if automatic and not automatic_flush_enabled(): + return 0 + if args.reason == "session-end": + record_stop_fn(store, hook_input) + if os.environ.get("MEM0_CODE_SYNC_FLUSH") == "1": + print(json.dumps(checkpoint_session(store, hook_input, args.reason))) + else: + session_id = str(hook_input.get("session_id") or "unknown-session") + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + already_running = store.has_inflight_flush(repo.identity, session_id) + if already_running and args.reason == "session-end": + hand_off_flush(hook_input, args.reason, wait_for_inflight=True) + elif not already_running and store.prepare_flush( + repo, session_id, args.reason, + ) is not None: + hand_off_flush(hook_input, args.reason) + elif extra_actions and args.action in extra_actions: + result = extra_actions[args.action](store, hook_input) + if result: + print(json.dumps(result)) + finally: + store.close() + return 0 + + +def entry_point( + *, + record_stop_fn=None, + extra_actions: dict | None = None, + data_dir_env: str = "MEM0_PLUGIN_DATA_DIR", + automatic_flush_reasons: set | None = None, +) -> None: + try: + raise SystemExit(run( + record_stop_fn=record_stop_fn, + extra_actions=extra_actions, + data_dir_env=data_dir_env, + automatic_flush_reasons=automatic_flush_reasons, + )) + except Exception as exc: + log_failure(exc) + raise SystemExit(0) + + +if __name__ == "__main__": + entry_point() diff --git a/integrations/kimi-plugin/core/mcp_server.py b/integrations/kimi-plugin/core/mcp_server.py new file mode 100644 index 000000000..036fbbdc9 --- /dev/null +++ b/integrations/kimi-plugin/core/mcp_server.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Expose Mem0's memory search as one local coding-agent tool.""" + +from __future__ import annotations + +import json +import os +import sys +from typing import Any + +import telemetry +from memory_core import ( + CODING_MEMORY_CATEGORY_NAMES, + PLUGIN_VERSION, + SEARCH_SCOPES, + format_search_result, + resolve_repo, + search_memories, +) + +PROTOCOL_VERSION = "2024-11-05" +TOOL_NAME = "search_memories" +TOOL_DESCRIPTION = ( + "Search memories from earlier work in this repository. ALWAYS call this " + "tool before answering anything that could depend on prior context: the " + "user's preferences, facts about this codebase, history, people, projects, " + "or earlier decisions. Do not rely on the chat window alone. The " + "repository's memory is shared by everyone who works in it and includes " + "what it took to run, test, or build here, so search before assuming an " + "invocation works. The scope argument changes what is searched: 'repo' " + "(default) is the whole repository's shared memory plus your own " + "preferences, 'dir' narrows the shared part to the directory you are " + "working in, and 'mine' is your preferences alone." +) +TOOL_SCHEMA = { + "type": "object", + "properties": { + "query": { + "type": "string", + "minLength": 1, + "maxLength": 2000, + "description": "A direct question about earlier work in this repository.", + }, + "top_k": { + "type": "integer", + "minimum": 1, + "maximum": 20, + "description": "Maximum memories to return. Uses Mem0's configured default when omitted.", + }, + "category": { + "type": "string", + "enum": list(CODING_MEMORY_CATEGORY_NAMES), + "description": "Optional memory category. Omit to search every category.", + }, + "scope": { + "type": "string", + "enum": list(SEARCH_SCOPES), + "description": ( + "Which memories to search. 'repo' (default) is the whole repository's " + "shared memory plus your own preferences, 'dir' narrows the shared " + "part to the current directory, 'mine' is your preferences alone." + ), + }, + "run_id": { + "type": "string", + "minLength": 1, + "description": ( + "Optional coding-agent session ID. With any scope, restricts results to memories " + "saved in that session. Omit to recall memories across sessions." + ), + }, + }, + "required": ["query"], + "additionalProperties": False, +} + + +class ToolInputError(ValueError): + pass + + +def _validate_arguments( + arguments: Any, +) -> tuple[str, int | None, str | None, str | None, str | None]: + if not isinstance(arguments, dict): + raise ToolInputError("Search arguments must be an object.") + + unknown = set(arguments) - {"query", "top_k", "category", "scope", "run_id"} + if unknown: + raise ToolInputError(f"Unknown search argument: {sorted(unknown)[0]}") + + query = arguments.get("query") + if not isinstance(query, str) or not query.strip(): + raise ToolInputError("query must be a non-empty string.") + query = query.strip() + if len(query) > 2000: + raise ToolInputError("query must be at most 2,000 characters.") + + top_k = arguments.get("top_k") + if top_k is not None and ( + isinstance(top_k, bool) or not isinstance(top_k, int) or not 1 <= top_k <= 20 + ): + raise ToolInputError("top_k must be an integer from 1 to 20.") + + category = arguments.get("category") + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ToolInputError("category must be one of Mem0's supported categories.") + + scope = arguments.get("scope") + if scope is not None and scope not in SEARCH_SCOPES: + raise ToolInputError(f"scope must be one of {list(SEARCH_SCOPES)}.") + + run_id = arguments.get("run_id") + if run_id is not None: + if not isinstance(run_id, str) or not run_id.strip(): + raise ToolInputError("run_id must be a non-empty string.") + run_id = run_id.strip() + + return query, top_k, category, scope, run_id + + +def call_search_memories(arguments: Any, cwd: str | None = None) -> str: + query, top_k, category, scope, run_id = _validate_arguments(arguments) + repo = resolve_repo(cwd or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + result = search_memories( + None, + repo, + None, + query, + top_k=top_k, + category=category, + scope=scope, + run_id=run_id, + operation="mcp-search", + ) + return format_search_result(result) + + +def _workspace_cwd(params: dict[str, Any]) -> str | None: + meta = params.get("_meta") + if not isinstance(meta, dict): + return None + metadata = meta.get("x-codex-turn-metadata") + if not isinstance(metadata, dict): + return None + workspaces = metadata.get("workspaces") or {} + if isinstance(workspaces, dict): + return next((path for path in workspaces if isinstance(path, str) and path), None) + return None + + +def _tool_response(text: str, *, is_error: bool = False) -> dict[str, Any]: + return { + "content": [{"type": "text", "text": text}], + "isError": is_error, + } + + +def handle_request(message: Any) -> dict[str, Any] | None: + if not isinstance(message, dict): + return None + request_id = message.get("id") + method = message.get("method") + + if method == "notifications/initialized": + return None + if method == "initialize": + requested = (message.get("params") or {}).get("protocolVersion") + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "protocolVersion": requested or PROTOCOL_VERSION, + "capabilities": {"tools": {"listChanged": False}}, + "serverInfo": {"name": "mem0", "version": PLUGIN_VERSION}, + }, + } + if method == "ping": + return {"jsonrpc": "2.0", "id": request_id, "result": {}} + if method == "tools/list": + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "tools": [ + { + "name": TOOL_NAME, + "description": TOOL_DESCRIPTION, + "inputSchema": TOOL_SCHEMA, + "annotations": { + "readOnlyHint": True, + "idempotentHint": True, + "openWorldHint": True, + }, + } + ] + }, + } + if method == "tools/call": + params = message.get("params") or {} + if params.get("name") != TOOL_NAME: + result = _tool_response("Unknown Mem0 tool.", is_error=True) + else: + try: + result = _tool_response( + call_search_memories(params.get("arguments"), _workspace_cwd(params)) + ) + except ToolInputError as exc: + result = _tool_response(str(exc), is_error=True) + except Exception: + result = _tool_response("Memory search failed.", is_error=True) + return {"jsonrpc": "2.0", "id": request_id, "result": result} + if request_id is None: + return None + return { + "jsonrpc": "2.0", + "id": request_id, + "error": {"code": -32601, "message": "Method not found"}, + } + + +def main() -> int: + for raw_line in sys.stdin: + try: + message = json.loads(raw_line) + response = handle_request(message) + except json.JSONDecodeError: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32700, "message": "Parse error"}, + } + except Exception: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32603, "message": "Internal error"}, + } + if response is not None: + sys.stdout.write(json.dumps(response, separators=(",", ":")) + "\n") + sys.stdout.flush() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/kimi-plugin/core/memory_cli.py b/integrations/kimi-plugin/core/memory_cli.py new file mode 100644 index 000000000..595729feb --- /dev/null +++ b/integrations/kimi-plugin/core/memory_cli.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Mem0 diagnostics and user controls.""" + +from __future__ import annotations + +import argparse +import json +import os + +import telemetry +from memory_core import ( + EvidenceStore, + api_key, + data_dir, + doctor, + forget_remote_repo, + configure_harness, + resolve_repo, + user_id, +) + + +def _print_status(value: dict) -> None: + last = value.get("last_operation") or {} + print(f"Mem0: {'paused' if value['paused'] else 'active'}") + print(f"Repository: {value['repo_id']}") + print(f"Local data: {value['data_dir']}") + print(f"API key: {'configured' if value['api_key_configured'] else 'missing'}") + print( + "Saved on this computer: " + f"{value['events']} session details, {value['flushes']} memory updates" + ) + print( + f"Used in this repository: {value['retrievals']} memories returned, " + f"{value['sidekick_runs']} sidekick runs" + ) + if last: + item_label = "" + if last["operation"] in {"flush", "flush-retry"}: + item_label = f", {last['item_count']} memories" + operation = ( + "memory update" + if last["operation"] in {"flush", "flush-retry"} + else last["operation"].replace("-", " ") + ) + print( + f"Last {operation}: " + f"{'succeeded' if last['success'] else 'failed'} " + f"({last['duration_ms']:.1f} ms{item_label})" + ) + sidekick = value.get("last_sidekick") or {} + if sidekick: + state = "finished" if sidekick.get("stopped_at") else "started" + print( + "Last sidekick: " + f"{state}, received {sidekick['context_chars']} characters of memory, " + f"agent {sidekick['agent_id']}" + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + subparsers = parser.add_subparsers(dest="command", required=True) + + status = subparsers.add_parser("status") + status.add_argument("--json", action="store_true") + + doctor_parser = subparsers.add_parser("doctor") + doctor_parser.add_argument("--json", action="store_true") + + subparsers.add_parser("pause") + subparsers.add_parser("resume") + + forget = subparsers.add_parser("forget") + forget.add_argument("--remote", action="store_true") + forget.add_argument("--yes", action="store_true") + forget.add_argument("--include-project-memory", action="store_true") + + args = parser.parse_args() + if args.harness: + source_tag = f"{args.harness.replace('-', '_')}_plugin" + configure_harness(args.harness, source_tag=source_tag) + telemetry.init(harness=args.harness, source_tag=source_tag.upper()) + if args.plugin_data_dir: + os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir + store = EvidenceStore() + try: + repo = resolve_repo(os.getcwd()) + telemetry.record("control", repo=repo, action=args.command) + if args.command == "status": + result = { + **store.status(repo.identity), + "repo_id": repo.identity, + "app_id": repo.app_id, + "project_id": repo.project_id, + "directory": repo.directory, + "user_id": user_id(), + "data_dir": str(data_dir()), + "api_key_configured": bool(api_key()), + } + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + _print_status(result) + elif args.command == "doctor": + result = doctor(os.getcwd()) + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + for name, check in result["checks"].items(): + print( + f"{'PASS' if check['ok'] else 'FAIL'} {name}: {check['detail']}" + ) + return 0 if result["ok"] else 1 + elif args.command == "pause": + store.set_setting("paused", "true") + print("Mem0 stopped saving and searching memories.") + elif args.command == "resume": + store.set_setting("paused", "false") + print("Mem0 resumed saving and searching memories.") + elif args.command == "forget": + if not args.yes: + print( + "Refusing to delete data without --yes. Add --remote to also " + "delete this user/repository scope from Mem0." + ) + return 2 + remote_result = ( + forget_remote_repo( + repo, include_project_memory=args.include_project_memory + ) + if args.remote + else None + ) + local_result = store.forget_local_repo(repo.identity) + print( + json.dumps( + {"local": local_result, "remote": remote_result}, + indent=2, + default=str, + ) + ) + if remote_result and remote_result.get("status") == "error": + return 1 + finally: + store.close() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/kimi-plugin/core/memory_core.py b/integrations/kimi-plugin/core/memory_core.py new file mode 100644 index 000000000..cf71196b8 --- /dev/null +++ b/integrations/kimi-plugin/core/memory_core.py @@ -0,0 +1,2645 @@ +#!/usr/bin/env python3 +"""Shared core for Mem0 agent plugins. + +Hooks record small session details locally. When the agent compacts or ends the +session, Mem0 sends the useful parts to the platform so it can create memories. +The agent can search those memories during later work in the repository. +""" + +from __future__ import annotations + +import functools +import hashlib +import json +import math +import os +import re +import sqlite3 +import subprocess +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +import telemetry + +DEFAULT_API_URL = "https://api.mem0.ai" +PLUGIN_VERSION = "0.3.1" + +_harness_name: str = "generic" +_harness_env_prefix: str = "MEM0_PLUGIN" +_harness_data_dir_name: str = "mem0-plugin" +_harness_source_tag: str = "mem0_plugin" + + +def configure_harness( + name: str, + env_prefix: str = "", + data_dir_name: str = "", + source_tag: str = "", +) -> None: + global _harness_name, _harness_env_prefix, _harness_data_dir_name, _harness_source_tag + _harness_name = name + _harness_env_prefix = env_prefix or f"MEM0_{name.upper().replace('-', '_')}" + _harness_data_dir_name = data_dir_name or f"{name}-plugin" + _harness_source_tag = source_tag or f"{name.replace('-', '_')}_plugin" + + +def harness_config() -> dict[str, str]: + return { + "name": _harness_name, + "env_prefix": _harness_env_prefix, + "data_dir_name": _harness_data_dir_name, + "source_tag": _harness_source_tag, + } + + +MAX_COMMAND_CHARS = 2000 +MAX_RESULT_CHARS = 2500 +MAX_EPISODE_CHARS = 12000 +CHECKPOINT_EXCHANGES = 5 +CHECKPOINT_MESSAGES = 10 +CHECKPOINT_SOURCE_CHARS = 40000 +DEFAULT_MAX_CONTEXT_CHARS = 4000 +MAX_EXTRACTION_INPUT_TOKENS = 24000 +MAX_FLUSH_ATTEMPTS = 5 +FORGET_PAGE_SIZE = 100 +FORGET_MAX_PAGES = 50 + +PROJECT_MEMORY_INSTRUCTIONS = """Save concise repository facts that will help anyone with future coding work in this repository. + +A completed change should produce one memory explaining the resulting behavior, where it is implemented when useful, and any important constraints or reasoning. Exploration or accepted decisions may produce separate memories only when they are independently useful. + +A command that failed and was then made to work should produce one memory naming the failing invocation, the error it returned, and the invocation that succeeded. Do not save one-off errors caused by an edit still in progress, transient network failures, or anything a rerun would fix on its own. + +Use the current coding agent's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or the coding agent completed them. Treat subagent responses as supporting repository evidence, not as decisions. + +Write about the repository, not the user, assistant, session, or task. Do not save personal preferences. Do not save a memory that only states which repository, branch, or directory the session worked in. Do not include test results, documentation updates, release notes, or temporary state. + +If nothing useful was established, return no memories.""" + +PERSONAL_MEMORY_INSTRUCTIONS = """Save concise facts about the user that will help in any repository: preferred tools, package managers, languages, coding style, review and communication preferences, and anything the user explicitly asked to be remembered about themselves. + +Write in the third person about the user, not about the repository, the assistant, the session, or the task. Do not save repository facts, project decisions, commands, or what was built. + +Never save that the user has no preferences or that nothing was learned. If nothing was learned about the user, return no memories.""" + +CODING_MEMORY_CATEGORIES = [ + { + "project_knowledge": ( + "What the project is and how its code, APIs, data, files, and " + "components work." + ) + }, + { + "decisions_and_constraints": ( + "Why an approach was chosen, what must remain true, and rules future " + "work must follow." + ) + }, + { + "workflows": ( + "How to run, test, debug, deploy, configure, or otherwise work on the " + "project." + ) + }, + { + "problems_and_fixes": ( + "Bugs, failures, known pitfalls, their causes, and how to fix or avoid " + "them." + ) + }, + { + "results": ( + "Outcomes and measurements from tests, benchmarks, experiments, or " + "investigations." + ) + }, +] +CODING_MEMORY_CATEGORY_NAMES = tuple( + category_name + for category in CODING_MEMORY_CATEGORIES + for category_name in category +) + +TEST_COMMAND_RE = re.compile( + r"(?:^|\s)(?:pytest|py\.test|jest|vitest|go\s+test|cargo\s+test|" + r"npm\s+(?:run\s+)?test|pnpm\s+(?:run\s+)?test|yarn\s+test|" + r"mvn\s+test|gradle\s+test|make\s+test)(?:\s|$)", + re.IGNORECASE, +) +BUILD_COMMAND_RE = re.compile( + r"(?:^|\s)(?:npm|pnpm|yarn)\s+(?:run\s+)?build(?:\s|$)|" + r"(?:^|\s)(?:cargo|go|mvn|gradle|make)\s+build(?:\s|$)", + re.IGNORECASE, +) + +SECRET_PATTERNS = [ + re.compile(r"(?i)(authorization\s*[:=]\s*(?:bearer|token)\s+)[^\s\"']+"), + re.compile( + r"(?i)((?:api[_-]?key|secret[_-]?access[_-]?key|session[_-]?token)\s*[:=]\s*)[^\s\"']+" + ), + re.compile( + r"(?i)((?:access[_-]?token|refresh[_-]?token|password|credential)" + r"\s*[:=]\s*)[^\s&\"']+" + ), + re.compile(r"\b(?:sk|m0|mem0_sk|psk)-[A-Za-z0-9_\-]{12,}\b"), + re.compile(r"\b(?:ASIA|AKIA)[A-Z0-9]{12,}\b"), + re.compile(r"\b(?:ghp_|github_pat_|xox[baprs]-)[A-Za-z0-9_\-]{12,}\b"), + re.compile( + r"-----BEGIN [^-]*PRIVATE KEY-----.*?-----END [^-]*PRIVATE KEY-----", + re.DOTALL, + ), + re.compile( + r'(?i)("(?:api[_-]?key|password|secret(?:[_-]?access[_-]?key)?' + r'|(?:access|refresh|session)[_-]?token|token|authorization|credential' + r')"\s*:\s*")(?:\\.|[^"\\])*' + ), +] + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def redact(value: Any) -> str: + text = ( + value + if isinstance(value, str) + else json.dumps(value, ensure_ascii=False, default=str) + ) + for pattern in SECRET_PATTERNS: + if pattern.groups: + text = pattern.sub(r"\1[REDACTED]", text) + else: + text = pattern.sub("[REDACTED]", text) + return text + + +def bounded(value: Any, limit: int) -> str: + text = redact(value).strip() + if len(text) <= limit: + return text + return text[:limit] + f"\n...[truncated {len(text) - limit} chars]" + + +def _git(cwd: str, *args: str) -> str: + try: + result = subprocess.run( + ["git", "-C", cwd, *args], + check=False, + capture_output=True, + text=True, + timeout=0.5, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return result.stdout.strip() if result.returncode == 0 else "" + + +def _normalize_remote(remote: str) -> str: + remote = remote.strip() + if remote.startswith("git@") and ":" in remote: + host_path = remote[4:].replace(":", "/", 1) + remote = f"https://{host_path}" + if remote.endswith(".git"): + remote = remote[:-4] + if "://" in remote: + parsed = urllib.parse.urlsplit(remote) + hostname = parsed.hostname or "" + if parsed.port: + hostname = f"{hostname}:{parsed.port}" + remote = urllib.parse.urlunsplit( + (parsed.scheme, hostname, parsed.path, parsed.query, parsed.fragment) + ) + return remote.rstrip("/") + + +_WILDCARD_SCOPE = re.compile(r"^\*+$") + + +def _scope_value(raw: str | None) -> str: + """Reject wildcards as identities: they are filter syntax and would widen the scope.""" + value = (raw or "").strip() + return "" if _WILDCARD_SCOPE.match(value) else value + + +SEARCH_SCOPES = ("repo", "dir", "mine") +DEFAULT_SEARCH_SCOPE = "repo" + +def directory_app_id(repo: RepoContext) -> str: + """The app_id of the directory this session runs in: the repository at the root, repository/path below it.""" + return f"{repo.app_id}/{repo.directory}" if repo.directory else repo.app_id + + +def directory_chain(repo: RepoContext) -> list[str]: + """Every directory a memory belongs to, from the top-level folder down to the one it was written in.""" + parts = repo.directory.split("/") if repo.directory else [] + return ["/".join(parts[: index + 1]) for index in range(len(parts))] + + +def _shared_project_ids(repo: RepoContext) -> list[str]: + """Current and pre-upgrade namespaces, shared by recall and explicit deletion.""" + if not repo.identity.startswith("local:") and repo.project_id != repo.app_id: + return [repo.project_id, repo.app_id] + return [repo.project_id] + + +def _search_filters(user: str, repo: RepoContext, scope: str) -> dict[str, Any]: + """Build the scope filter: app_id scopes to the repo, then union shared and personal lanes.""" + app_scope = {"app_id": repo.app_id} + mine = {"AND": [{"user_id": user}, app_scope]} + if scope == "mine": + return mine + projects = [{"AND": [{"agent_id": project_id}, app_scope]} for project_id in _shared_project_ids(repo)] + shared: dict[str, Any] = projects[0] if len(projects) == 1 else {"OR": projects} + if scope == "dir" and repo.directory: + shared = {"AND": [shared, {"metadata": {"dirs": {"contains": repo.directory}}}]} + return {"OR": [shared, mine]} + + +def search_scope() -> str: + configured = ( + _plugin_option("search_scope", "MEM0_CODE_SEARCH_SCOPE") or "" + ).strip().lower() + return configured if configured in SEARCH_SCOPES else DEFAULT_SEARCH_SCOPE + + +def resolve_search_scope(scope: str | None) -> str: + value = (scope or search_scope()).strip().lower() + if value not in SEARCH_SCOPES: + raise ValueError(f"Unknown search scope: {value}") + return value + + +def _legacy_project_map(cwd: str, root: str, raw_remote: str) -> str: + """Return the project name used by the previous Claude Code plugin.""" + try: + data = json.loads((Path.home() / ".mem0" / "project_map.json").read_text()) + except (OSError, json.JSONDecodeError): + return "" + if not isinstance(data, dict): + return "" + + keys = list(dict.fromkeys([cwd, root, os.path.realpath(cwd), os.path.realpath(root)])) + if raw_remote: + keys.append(f"remote:{hashlib.sha256(raw_remote.encode()).hexdigest()[:16]}") + for key in keys: + value = data.get(key) + if isinstance(value, str) and value.strip(): + return value.strip() + return "" + + +def _legacy_project_id(cwd: str, root: str, raw_remote: str, identity: str) -> str: + """Use the repository namespace created by the previous Mem0 plugin.""" + configured = _scope_value(os.environ.get("MEM0_PROJECT_ID")) + if configured: + return configured + + mapped = _scope_value(_legacy_project_map(cwd, root, raw_remote)) + if mapped: + return mapped + + remote = raw_remote or ("" if identity.startswith("local:") else identity) + remote = remote.strip().removesuffix(".git") + for prefix in ("https://", "http://", "ssh://", "git://"): + if remote.startswith(prefix): + remote = remote[len(prefix) :] + break + else: + remote = re.sub(r"^git@", "", remote) + parts = [part for part in remote.replace(":", "/", 1).split("/") if part] + if len(parts) >= 2: + return f"{parts[-2]}-{parts[-1]}".replace("/", "-").replace(":", "-") + if parts: + return parts[-1].replace("/", "-").replace(":", "-") + return os.path.basename(root or cwd) or "unknown" + + +@dataclass(frozen=True) +class RepoContext: + cwd: str + root: str + identity: str + app_id: str + branch: str + head_sha: str + project_id: str = "" + directory: str = "" + + +def _project_id(root: str, identity: str, app_id: str) -> str: + """The shared namespace: includes a host hash so repos with the same owner/name on different hosts stay apart.""" + if not identity.startswith("local:"): + return f"{app_id}-{hashlib.sha256(identity.encode()).hexdigest()[:10]}" + return f"local-{app_id}-{hashlib.sha256(root.encode()).hexdigest()[:10]}" + + +def _relative_directory(cwd: str, root: str) -> str: + relative = os.path.relpath(cwd, root) + return "" if relative == "." or relative.startswith("..") else relative.replace(os.sep, "/") + + +@dataclass(frozen=True) +class MemorySearchResult: + succeeded: bool + matched_count: int + already_shown_count: int + memories: list[dict[str, Any]] + + +@functools.lru_cache(maxsize=64) +def _resolve_repo_cached(cwd: str) -> RepoContext: + given_cwd = cwd + cwd = os.path.realpath(cwd) + given_root = _git(cwd, "rev-parse", "--show-toplevel") or given_cwd + root = os.path.realpath(given_root) + raw_remote = _git(root, "config", "--get", "remote.origin.url") + remote = _normalize_remote(raw_remote) + identity = remote or f"local:{root}" + app_id = _legacy_project_id(given_cwd, given_root, raw_remote, identity) + return RepoContext( + cwd=cwd, + root=root, + identity=identity, + app_id=app_id, + branch=_git(root, "branch", "--show-current") or "detached", + head_sha=_git(root, "rev-parse", "HEAD"), + project_id=_project_id(root, identity, app_id), + directory=_relative_directory(cwd, root), + ) + + +def resolve_repo(cwd: str | None) -> RepoContext: + return _resolve_repo_cached(os.path.abspath(cwd or os.getcwd())) + + +def api_key() -> str: + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return configured + try: + return (data_dir() / "api-key").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def cache_plugin_api_key() -> bool: + """Bridge host's hook-only sensitive config into plugin-owned storage.""" + configured = ( + os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if not configured: + return False + + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + path = directory / "api-key" + temporary = directory / f"api-key.{os.getpid()}.tmp" + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_TRUNC, + 0o600, + ) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + handle.write(configured) + os.replace(temporary, path) + os.chmod(path, 0o600) + finally: + try: + temporary.unlink() + except FileNotFoundError: + pass + return True + + +def clear_stale_api_key_cache() -> bool: + """Drop the cached key file once every configured key source is gone.""" + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return False + path = data_dir() / "api-key" + if not path.exists(): + return False + try: + path.unlink() + except OSError: + return False + return True + + +def detached_process_kwargs(platform: str | None = None) -> dict: + """Keep a spawned worker alive after the coding agent exits, on POSIX and Windows.""" + if (platform or sys.platform) == "win32": + return { + "creationflags": subprocess.DETACHED_PROCESS + | subprocess.CREATE_NEW_PROCESS_GROUP + } + return {"start_new_session": True} + + +def _plugin_option(name: str, fallback: str = "") -> str: + return ( + os.environ.get(f"PLUGIN_OPTION_{name.upper()}") + or os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") + or os.environ.get(fallback) + or "" + ).strip() + + +def user_id() -> str: + return ( + _scope_value(_plugin_option("user_id", "MEM0_CODE_USER_ID")) + or _scope_value(os.environ.get("MEM0_USER_ID")) + or _scope_value(os.environ.get("MEM0_RESOLVED_USER_ID")) + or _scope_value(os.environ.get("USER")) + or _scope_value(os.environ.get("USERNAME")) + or "default" + ) + + +def data_dir() -> Path: + configured = ( + os.environ.get("MEM0_CODE_DATA_DIR") + or os.environ.get("MEM0_PLUGIN_DATA_DIR") + or os.environ.get("PLUGIN_DATA") + or os.environ.get("CLAUDE_PLUGIN_DATA") + ) + return ( + Path(configured).expanduser() if configured else Path.home() / ".mem0" / _harness_data_dir_name + ) + + +def _bool_option(name: str, fallback: str, default: bool = False) -> bool: + value = _plugin_option(name, fallback) + if not value: + return default + return value.lower() in {"1", "true", "yes", "on"} + + +def _int_option(name: str, fallback: str, default: int) -> int: + value = _plugin_option(name, fallback) + try: + return int(value) if value else default + except ValueError: + return default + + + +def _checkpoint_message(event: dict[str, Any]) -> str: + kind = event.get("kind") + payload = event.get("payload") or {} + if kind == "user_prompt": + return redact(payload.get("text", "")).strip() + if kind == "assistant_stop": + transcript_messages = payload.get("transcript_messages") or [] + if isinstance(transcript_messages, list): + text = "\n".join( + str(message.get("content") or "") + for message in transcript_messages + if isinstance(message, dict) and message.get("content") + ) + if text: + return text + return redact(payload.get("text", "")).strip() + if kind == "sidekick_stop": + return redact(payload.get("final_message", "")).strip() + return "" + + +def checkpoint_stats(events: list[dict[str, Any]]) -> tuple[int, int, int]: + """Return completed exchanges, messages, and source characters.""" + completed = sum(event.get("kind") == "assistant_stop" for event in events) + contents = [content for event in events if (content := _checkpoint_message(event))] + return completed, len(contents), sum(len(content) for content in contents) + + +def select_checkpoint_events( + events: list[dict[str, Any]], *, force: bool +) -> list[dict[str, Any]]: + """Select one ordered extraction block without splitting an exchange.""" + for index, event in enumerate(events): + if event.get("kind") != "assistant_stop": + continue + candidate = events[: index + 1] + completed, messages, source_chars = checkpoint_stats(candidate) + if ( + completed >= CHECKPOINT_EXCHANGES + or messages >= CHECKPOINT_MESSAGES + or source_chars >= CHECKPOINT_SOURCE_CHARS + ): + return candidate + return events if force else [] + + +class EvidenceStore: + def __init__(self, path: Path | None = None): + directory = data_dir() if path is None else path.parent + directory.mkdir(parents=True, exist_ok=True) + self.path = path or directory / "evidence.sqlite3" + try: + self._open() + except sqlite3.DatabaseError: + self._quarantine() + self._open() + + def _open(self) -> None: + self.conn = sqlite3.connect(self.path, timeout=10) + self.conn.row_factory = sqlite3.Row + try: + self.conn.execute("PRAGMA journal_mode=WAL") + self.conn.execute("PRAGMA busy_timeout=10000") + self._migrate() + except sqlite3.DatabaseError: + self.conn.close() + raise + + def _quarantine(self) -> None: + """Move an unreadable database aside so capture restarts cleanly.""" + stamp = int(time.time()) + for suffix in ("", "-wal", "-shm"): + source = Path(f"{self.path}{suffix}") + try: + source.replace(f"{self.path}.corrupt-{stamp}{suffix}") + except FileNotFoundError: + continue + except OSError: + try: + source.unlink() + except OSError: + pass + telemetry.record("db_quarantined") + + def close(self) -> None: + self.conn.close() + + def _migrate(self) -> None: + self.conn.executescript( + """ + CREATE TABLE IF NOT EXISTS events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + created_at TEXT NOT NULL, + kind TEXT NOT NULL, + payload_json TEXT NOT NULL, + flush_id TEXT + ); + CREATE INDEX IF NOT EXISTS events_session_idx + ON events(repo_id, session_id, flush_id, id); + + CREATE TABLE IF NOT EXISTS session_scopes ( + session_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + root TEXT NOT NULL, + branch TEXT NOT NULL, + head_sha TEXT NOT NULL, + created_at TEXT NOT NULL, + directory TEXT NOT NULL DEFAULT '' + ); + + CREATE TABLE IF NOT EXISTS flushes ( + packet_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + reason TEXT NOT NULL, + event_start INTEGER NOT NULL, + event_end INTEGER NOT NULL, + status TEXT NOT NULL, + episode_event_id TEXT, + semantic_event_id TEXT, + error TEXT, + attempts INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + CREATE TABLE IF NOT EXISTS retrievals ( + session_id TEXT NOT NULL, + repo_id TEXT NOT NULL, + memory_id TEXT NOT NULL, + injected_at TEXT NOT NULL, + rank INTEGER, + score REAL, + memory_text TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(session_id, repo_id, memory_id) + ); + + CREATE TABLE IF NOT EXISTS operations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + created_at TEXT NOT NULL, + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + operation TEXT NOT NULL, + duration_ms REAL NOT NULL, + success INTEGER NOT NULL, + item_count INTEGER NOT NULL DEFAULT 0, + request_chars INTEGER NOT NULL DEFAULT 0, + response_chars INTEGER NOT NULL DEFAULT 0, + error TEXT + ); + + CREATE TABLE IF NOT EXISTS sidekick_runs ( + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + agent_type TEXT NOT NULL, + started_at TEXT NOT NULL, + stopped_at TEXT, + transcript_path TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + final_message TEXT, + PRIMARY KEY(repo_id, session_id, agent_id) + ); + CREATE INDEX IF NOT EXISTS sidekick_runs_repo_idx + ON sidekick_runs(repo_id, started_at); + + CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + """ + ) + # Remove the pre-0.1.1 no-tools snapshot implementation. The real coding + # sidekick is a native Claude Code agent and stores no state in this DB. + self.conn.executescript( + """ + DROP TABLE IF EXISTS sidekick_calls; + DROP TABLE IF EXISTS sidekick_snapshots; + DROP TABLE IF EXISTS sidekick_state; + DROP TABLE IF EXISTS sidekick_packets; + """ + ) + self._ensure_column("retrievals", "rank", "INTEGER") + self._ensure_column("retrievals", "score", "REAL") + self._ensure_column("retrievals", "memory_text", "TEXT") + self._ensure_column("retrievals", "context_chars", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("flushes", "attempts", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("session_scopes", "directory", "TEXT NOT NULL DEFAULT ''") + self.conn.commit() + + def _ensure_column(self, table: str, column: str, declaration: str) -> None: + columns = { + str(row["name"]) + for row in self.conn.execute(f"PRAGMA table_info({table})").fetchall() + } + if column not in columns: + self.conn.execute(f"ALTER TABLE {table} ADD COLUMN {column} {declaration}") + + def record_event( + self, + repo: RepoContext, + session_id: str, + kind: str, + payload: dict[str, Any], + ) -> int: + cursor = self.conn.execute( + """INSERT INTO events + (repo_id, app_id, session_id, created_at, kind, payload_json) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + repo.app_id, + session_id, + utc_now(), + kind, + json.dumps(payload, ensure_ascii=False, sort_keys=True), + ), + ) + self.conn.commit() + return int(cursor.lastrowid) + + def record_assistant_response(self, repo: RepoContext, session_id: str, message: str) -> None: + """Ignore repeated response hooks until another prompt or a different answer arrives.""" + with self.conn: + # Serialize the check and insert across concurrent Stop and SessionEnd hooks. + self.conn.execute("BEGIN IMMEDIATE") + previous = self.conn.execute( + """SELECT kind, payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind IN ('user_prompt', 'assistant_stop') + ORDER BY id DESC LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if ( + previous is not None + and previous["kind"] == "assistant_stop" + and json.loads(previous["payload_json"]).get("text") == message + ): + return + self.record_event(repo, session_id, "assistant_stop", {"text": message}) + + def repo_for_session(self, session_id: str, cwd: str | None) -> RepoContext: + """Keep one project scope for every hook in a coding-agent session.""" + current = resolve_repo(cwd) + if session_id == "unknown-session": + return current + + with self.conn: + self.conn.execute( + """INSERT OR IGNORE INTO session_scopes + (session_id, repo_id, app_id, root, branch, head_sha, created_at, directory) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + current.identity, + current.app_id, + current.root, + current.branch, + current.head_sha, + utc_now(), + current.directory, + ), + ) + scope = self.conn.execute( + "SELECT * FROM session_scopes WHERE session_id = ?", (session_id,) + ).fetchone() + same_git_repo = ( + current.identity == scope["repo_id"] and bool(current.head_sha) + ) + pinned = current if same_git_repo else resolve_repo(str(scope["root"])) + return RepoContext( + cwd=current.cwd, + root=pinned.root, + identity=str(scope["repo_id"]), + app_id=str(scope["app_id"]), + branch=pinned.branch, + head_sha=pinned.head_sha, + project_id=pinned.project_id, + directory=str(scope["directory"] or ""), + ) + + def prepare_flush( + self, repo: RepoContext, session_id: str, reason: str + ) -> tuple[str, list[dict[str, Any]]] | None: + with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + existing = self.conn.execute( + """SELECT * FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') + ORDER BY created_at LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: + self.conn.execute( + "UPDATE flushes SET status = 'gave-up', updated_at = ? WHERE packet_id = ?", + (utc_now(), existing["packet_id"]), + ) + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status="gave-up", + success=False, + attempts=int(existing["attempts"] or 0), + ) + existing = None + if existing: + if reason != "periodic" and existing["reason"] == "periodic": + self.conn.execute( + "UPDATE flushes SET reason = ?, updated_at = ? WHERE packet_id = ?", + (reason, utc_now(), existing["packet_id"]), + ) + existing_rows = self.conn.execute( + "SELECT * FROM events WHERE flush_id = ? ORDER BY id", + (existing["packet_id"],), + ).fetchall() + if existing_rows: + return str(existing["packet_id"]), [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in existing_rows + ] + + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo.identity, session_id), + ).fetchall() + if not rows: + return None + + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + events = select_checkpoint_events(events, force=reason != "periodic") + if not events: + return None + event_start, event_end = events[0]["id"], events[-1]["id"] + packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" + packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] + now = utc_now() + + self.conn.execute( + """INSERT OR IGNORE INTO flushes + (packet_id, repo_id, app_id, session_id, reason, event_start, + event_end, status, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, 'prepared', ?, ?)""", + ( + packet_id, + repo.identity, + repo.app_id, + session_id, + reason, + event_start, + event_end, + now, + now, + ), + ) + event_ids = [event["id"] for event in events] + placeholders = ", ".join("?" for _ in event_ids) + self.conn.execute( + f"UPDATE events SET flush_id = ? " + f"WHERE id IN ({placeholders}) AND flush_id IS NULL", + (packet_id, *event_ids), + ) + return packet_id, events + + def checkpoint_due(self, repo_id: str, session_id: str) -> bool: + if self.has_inflight_flush(repo_id, session_id): + return False + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo_id, session_id), + ).fetchall() + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + return bool(select_checkpoint_events(events, force=False)) + + def has_inflight_flush(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status IN ('prepared', 'semantic-queued') + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def flush_record(self, packet_id: str) -> dict[str, Any] | None: + row = self.conn.execute( + "SELECT * FROM flushes WHERE packet_id = ?", (packet_id,) + ).fetchone() + return dict(row) if row else None + + def has_unflushed_events(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def unflushed_starts_with_session_start( + self, repo_id: str, session_id: str + ) -> bool: + row = self.conn.execute( + """SELECT kind FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id LIMIT 1""", + (repo_id, session_id), + ).fetchone() + return bool(row and row["kind"] == "session_start") + + def update_flush(self, packet_id: str, **fields: Any) -> None: + allowed = {"status", "episode_event_id", "semantic_event_id", "error"} + updates = {key: value for key, value in fields.items() if key in allowed} + updates["updated_at"] = utc_now() + clause = ", ".join(f"{key} = ?" for key in updates) + failed = str(fields.get("status", "")) in { + "error", + "semantic-failed", + "semantic-timeout", + "semantic-missing", + } + if failed: + clause += ", attempts = attempts + 1" + with self.conn: + self.conn.execute( + f"UPDATE flushes SET {clause} WHERE packet_id = ?", + [*updates.values(), packet_id], + ) + + def unseen( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> list[dict[str, Any]]: + seen = { + row["memory_id"] + for row in self.conn.execute( + "SELECT memory_id FROM retrievals WHERE session_id = ? AND repo_id = ?", + (session_id, repo_id), + ) + } + return [memory for memory in memories if str(memory.get("id", "")) not in seen] + + def mark_injected( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> None: + now = utc_now() + with self.conn: + for rank, memory in enumerate(memories, start=1): + memory_id = str(memory.get("id", "")) + if memory_id: + memory_text = bounded( + memory.get("memory") or memory.get("text") or "", + 4000, + ) + try: + score = float(memory["score"]) + except (KeyError, TypeError, ValueError): + score = None + self.conn.execute( + """INSERT OR IGNORE INTO retrievals + (session_id, repo_id, memory_id, injected_at, rank, + score, memory_text, context_chars) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + repo_id, + memory_id, + now, + rank, + score, + memory_text, + len(memory_text), + ), + ) + + def injected_memories( + self, session_id: str, repo_id: str + ) -> list[dict[str, Any]]: + """Return the exact memories already supplied to the main conversation.""" + rows = self.conn.execute( + """SELECT memory_id, rank, score, memory_text + FROM retrievals + WHERE session_id = ? AND repo_id = ? + ORDER BY COALESCE(rank, 2147483647), injected_at, memory_id""", + (session_id, repo_id), + ).fetchall() + return [ + { + "id": row["memory_id"], + "memory": row["memory_text"], + "score": row["score"], + } + for row in rows + if row["memory_text"] + ] + + def start_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + context_chars: int, + ) -> bool: + """Record one native sidekick instance and whether context was first sent.""" + with self.conn: + cursor = self.conn.execute( + """INSERT OR IGNORE INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + context_chars) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + utc_now(), + context_chars, + ), + ) + return int(cursor.rowcount) > 0 + + def stop_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + transcript_path: str, + final_message: str, + ) -> str: + now = utc_now() + self.conn.execute("BEGIN IMMEDIATE") + try: + if not agent_id: + rows = self.conn.execute( + """SELECT agent_id FROM sidekick_runs + WHERE repo_id = ? AND session_id = ? AND agent_type = ? AND stopped_at IS NULL + LIMIT 2""", + (repo.identity, session_id, agent_type), + ).fetchall() + # Without a host ID, overlapping runs cannot be correlated reliably. + agent_id = rows[0]["agent_id"] if len(rows) == 1 else f"unknown-agent-{time.time_ns()}" + self.conn.execute( + """INSERT INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + stopped_at, transcript_path, final_message) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(repo_id, session_id, agent_id) DO UPDATE SET + stopped_at = excluded.stopped_at, + transcript_path = excluded.transcript_path, + final_message = excluded.final_message""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + now, + now, + bounded(transcript_path, 2000), + redact(final_message).strip(), + ), + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + return agent_id + + def operation( + self, + repo: RepoContext, + session_id: str, + operation: str, + duration_ms: float, + success: bool, + *, + item_count: int = 0, + request_chars: int = 0, + response_chars: int = 0, + error: str = "", + ) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO operations + (created_at, repo_id, session_id, operation, duration_ms, + success, item_count, request_chars, response_chars, error) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + ( + utc_now(), + repo.identity, + session_id, + operation, + duration_ms, + int(success), + item_count, + request_chars, + response_chars, + bounded(error, 1000), + ), + ) + + def has_operation(self, repo_id: str, session_id: str, operation: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM operations + WHERE repo_id = ? AND session_id = ? AND operation = ? + LIMIT 1""", + (repo_id, session_id, operation), + ).fetchone() + return row is not None + + def has_event(self, repo_id: str, session_id: str, kind: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + return row is not None + + def latest_event_payload( + self, repo_id: str, session_id: str, kind: str + ) -> dict[str, Any]: + row = self.conn.execute( + """SELECT payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + ORDER BY id DESC LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + if not row: + return {} + try: + payload = json.loads(row["payload_json"]) + except json.JSONDecodeError: + return {} + return payload if isinstance(payload, dict) else {} + + def setting(self, key: str, default: str = "") -> str: + row = self.conn.execute( + "SELECT value FROM settings WHERE key = ?", (key,) + ).fetchone() + return str(row["value"]) if row else default + + def set_setting(self, key: str, value: str) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO settings(key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at""", + (key, value, utc_now()), + ) + + def is_paused(self) -> bool: + return self.setting("paused", "false").lower() in { + "1", + "true", + "yes", + "on", + } + + def forget_local_repo(self, repo_id: str) -> dict[str, int]: + tables = { + "events": "repo_id", + "session_scopes": "repo_id", + "flushes": "repo_id", + "retrievals": "repo_id", + "operations": "repo_id", + "sidekick_runs": "repo_id", + } + removed: dict[str, int] = {} + with self.conn: + for table, column in tables.items(): + cursor = self.conn.execute( + f"DELETE FROM {table} WHERE {column} = ?", (repo_id,) + ) + removed[table] = max(int(cursor.rowcount), 0) + return removed + + def status(self, repo_id: str) -> dict[str, Any]: + def count(table: str) -> int: + return int( + self.conn.execute( + f"SELECT COUNT(*) FROM {table} WHERE repo_id = ?", (repo_id,) + ).fetchone()[0] + ) + + last_operation = self.conn.execute( + """SELECT created_at, operation, duration_ms, success, item_count, error + FROM operations WHERE repo_id = ? ORDER BY id DESC LIMIT 1""", + (repo_id,), + ).fetchone() + last_sidekick = self.conn.execute( + """SELECT session_id, agent_id, agent_type, started_at, stopped_at, + context_chars + FROM sidekick_runs WHERE repo_id = ? + ORDER BY started_at DESC LIMIT 1""", + (repo_id,), + ).fetchone() + return { + "paused": self.is_paused(), + "events": count("events"), + "flushes": count("flushes"), + "retrievals": count("retrievals"), + "sidekick_runs": count("sidekick_runs"), + "last_operation": dict(last_operation) if last_operation else None, + "last_sidekick": dict(last_sidekick) if last_sidekick else None, + } + + +def _session_id(hook_input: dict[str, Any]) -> str: + return str(hook_input.get("session_id") or "unknown-session") + + +def record_session_start(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + store.record_event( + repo, + session_id, + "session_start", + { + "source": hook_input.get("source", "startup"), + "model": bounded(hook_input.get("model", ""), 200), + "branch": repo.branch, + "head_sha": repo.head_sha, + }, + ) + telemetry.record( + "session_start", + repo=repo, + session_id=session_id, + trigger=bounded(str(hook_input.get("source", "startup")), 60), + model=bounded(hook_input.get("model", ""), 200), + api_key_configured=bool(api_key()), + is_git_repo=not repo.identity.startswith("local:"), + ) + + +def record_user_prompt( + store: EvidenceStore, hook_input: dict[str, Any] +) -> tuple[RepoContext, str, str, bool]: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prompt = redact(hook_input.get("prompt", "")).strip() + is_first_prompt = not store.has_event(repo.identity, session_id, "user_prompt") + store.record_event(repo, session_id, "user_prompt", {"text": prompt}) + return repo, session_id, prompt, is_first_prompt + + +def _tool_result_preview(response: Any) -> str: + if isinstance(response, dict): + selected = {} + for key in ( + "stdout", + "stderr", + "output", + "content", + "error", + "filePath", + "success", + "interrupted", + ): + if key in response: + selected[key] = response[key] + response = selected or {"keys": sorted(response.keys())[:20]} + return bounded(response, MAX_RESULT_CHARS) + + +def tool_payload(hook_input: dict[str, Any], *, failed: bool | None = False) -> dict[str, Any]: + name = str(hook_input.get("tool_name") or "unknown") + tool_input = hook_input.get("tool_input") or {} + if not isinstance(tool_input, dict): + tool_input = {} + payload: dict[str, Any] = { + "tool": name, + "failed": failed, + "duration_ms": hook_input.get("duration_ms"), + "agent_role": "sidekick" if hook_input.get("agent_id") else "main", + } + if hook_input.get("agent_id"): + payload["agent_id"] = bounded(hook_input["agent_id"], 200) + if hook_input.get("agent_type"): + payload["agent_type"] = bounded(hook_input["agent_type"], 200) + + if name in {"Read", "Write", "Edit", "MultiEdit", "NotebookEdit"}: + path = tool_input.get("file_path") or tool_input.get("notebook_path") + if path: + payload["path"] = bounded(path, 1000) + if name in {"Write", "Edit", "MultiEdit", "NotebookEdit"}: + payload["mutation_chars"] = sum( + len(str(tool_input.get(key, ""))) + for key in ("content", "new_string", "new_source", "edits") + ) + elif name == "Bash" or "command" in tool_input: + command = bounded(tool_input.get("command", ""), MAX_COMMAND_CHARS) + payload["command"] = command + payload["command_kind"] = ( + "test" + if TEST_COMMAND_RE.search(command) + else "build" + if BUILD_COMMAND_RE.search(command) + else "shell" + ) + response = ( + hook_input.get("error") if failed else hook_input.get("tool_response") + ) + payload["result_preview"] = _tool_result_preview(response) + elif name in {"Grep", "Glob", "WebSearch", "WebFetch"}: + for key in ("pattern", "path", "query", "url"): + if tool_input.get(key): + payload[key] = bounded(tool_input[key], 1000) + else: + payload["input_keys"] = sorted(tool_input.keys())[:20] + if failed: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + + if failed and "error" not in payload: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + return payload + + +def record_tool( + store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool | None = False +) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + payload = tool_payload(hook_input, failed=failed) + if payload.get("path"): + payload["repo_path"] = _repo_relative_path(repo, str(payload["path"])) + store.record_event( + repo, + session_id, + "tool_failure" if failed else "tool_result", + payload, + ) + + +def record_sidekick_start( + store: EvidenceStore, hook_input: dict[str, Any], *, inject_context: bool = True +) -> str: + """Record a native sidekick and reuse the main turn's retrieved memories.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_id = bounded(hook_input.get("agent_id", "unknown-agent"), 200) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + context = combine_context( + format_context(store.injected_memories(session_id, repo.identity)) + ) + if not inject_context: + context = "" + first_start = store.start_sidekick( + repo, session_id, agent_id, agent_type, len(context) + ) + store.record_event( + repo, + session_id, + "sidekick_start", + { + "agent_id": agent_id, + "agent_type": agent_type, + "context_chars": len(context) if first_start else 0, + "worktree_root": bounded(repo.root, 2000), + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="start", + first_start=first_start, + context_chars=len(context) if first_start else 0, + ) + return context if first_start else "" + + +def record_sidekick_stop(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + agent_id = bounded(hook_input.get("agent_id", ""), 200) + final_message = redact(hook_input.get("last_assistant_message", "")).strip() + transcript_path = bounded(hook_input.get("agent_transcript_path", ""), 2000) + agent_id = store.stop_sidekick( + repo, + session_id, + agent_id, + agent_type, + transcript_path, + final_message, + ) + store.record_event( + repo, + session_id, + "sidekick_stop", + { + "agent_id": agent_id, + "agent_type": agent_type, + "transcript_path": transcript_path, + "final_message": final_message, + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="stop", + has_transcript=bool(transcript_path), + message_chars=len(final_message), + ) + + +def _ordered_unique(values: Iterable[str]) -> list[str]: + seen: set[str] = set() + result = [] + for value in values: + if value and value not in seen: + seen.add(value) + result.append(value) + return result + + +def _repo_relative_path(repo: RepoContext, value: str) -> str: + value = str(value or "").strip() + if not value: + return "" + try: + path = Path(value) + if path.is_absolute(): + try: + return path.resolve().relative_to(Path(repo.root).resolve()).as_posix() + except ValueError: + return "" + except (OSError, ValueError): + pass + return bounded(value, 1000) + + +def _render_command_lines(commands: list[dict[str, str]]) -> list[str]: + lines = [] + for command in commands: + line = f"- [{command['status']}/{command['kind']}] {command['command']}" + if command["result"]: + line += f" — {bounded(command['result'], 500).replace(chr(10), ' ')}" + lines.append(line) + return lines + + +def build_episode( + repo: RepoContext, + session_id: str, + packet_id: str, + events: list[dict[str, Any]], + *, + canonical_task: str = "", + task_outcome: str = "", +) -> tuple[str, dict[str, Any]]: + prompts = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "user_prompt" and e["payload"].get("text") + ] + assistant_conclusions = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "assistant_stop" and e["payload"].get("text") + ] + sidekick_outcomes = [ + redact(e["payload"].get("final_message", "")).strip() + for e in events + if e["kind"] == "sidekick_stop" and e["payload"].get("final_message") + ] + tools = [ + e["payload"] for e in events if e["kind"] in {"tool_result", "tool_failure"} + and e["payload"].get("agent_role", "main") == "main" + ] + read_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") == "Read" + ) + modified_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") in {"Write", "Edit", "MultiEdit", "NotebookEdit"} + ) + searches = [ + {key: t[key] for key in ("tool", "pattern", "path", "query", "url") if key in t} + for t in tools + if t.get("tool") in {"Grep", "Glob", "WebSearch", "WebFetch"} + ] + commands = [ + { + "command": t.get("command", ""), + "kind": t.get("command_kind", "shell"), + "status": "unknown" if t.get("failed", False) is None else "failed" if t.get("failed") else "succeeded", + "result": t.get("result_preview", ""), + } + for t in tools + if t.get("command") + ] + + task = bounded(canonical_task or (prompts[0] if prompts else ""), 4000) + conclusion = redact(assistant_conclusions[-1] if assistant_conclusions else "").strip() + outcome = bounded(task_outcome, 2000) + + extraction_messages: list[dict[str, str]] = [] + pending_user_messages: list[dict[str, str]] = [] + if task and not prompts: + pending_user_messages.append({"role": "user", "content": task}) + for event in events: + if event["kind"] == "user_prompt" and event["payload"].get("text"): + pending_user_messages.append( + { + "role": "user", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + elif event["kind"] == "assistant_stop": + transcript_messages = event["payload"].get("transcript_messages") or [] + if isinstance(transcript_messages, list) and transcript_messages: + transcript_users = { + redact(message.get("content") or "").strip() + for message in transcript_messages + if isinstance(message, dict) and message.get("role") == "user" + } + extraction_messages.extend( + message + for message in pending_user_messages + if message["content"].strip() not in transcript_users + ) + extraction_messages.extend( + { + "role": str(message.get("role") or ""), + "content": redact(message.get("content") or "").strip(), + } + for message in transcript_messages + if isinstance(message, dict) + and message.get("role") in {"user", "assistant"} + and message.get("content") + ) + else: + extraction_messages.extend(pending_user_messages) + if event["payload"].get("text"): + extraction_messages.append( + { + "role": "assistant", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + pending_user_messages = [] + elif event["kind"] == "sidekick_stop": + pass + extraction_messages.extend(pending_user_messages) + + structured = { + "packet_id": packet_id, + "repo": repo.identity, + "app_id": repo.app_id, + "session_id": session_id, + "branch": repo.branch, + "head_sha": repo.head_sha, + "task": task, + "task_outcome": outcome, + "assistant_conclusion": conclusion, + "user_messages": prompts, + "assistant_outcomes": assistant_conclusions, + "sidekick_outcomes": sidekick_outcomes, + "extraction_messages": extraction_messages, + "files_read": read_paths[:50], + "files_modified": modified_paths[:50], + "searches": searches[-30:], + "commands": commands[-30:], + } + lines = ["Coding-session episode"] + if task: + lines.extend(["", "Task:", task]) + if modified_paths: + lines.extend( + ["", "Files modified:", *[f"- {path}" for path in modified_paths[:50]]] + ) + if read_paths: + lines.extend(["", "Files read:", *[f"- {path}" for path in read_paths[:50]]]) + if commands: + lines.append("") + lines.append("Observed commands:") + lines.extend(_render_command_lines(commands[-30:])) + if searches: + lines.extend( + [ + "", + "Observed searches:", + *[ + f"- {json.dumps(item, ensure_ascii=False, sort_keys=True)}" + for item in searches[-20:] + ], + ] + ) + if conclusion: + lines.extend(["", "Agent conclusion:", conclusion]) + if outcome: + lines.extend(["", "Task outcome:", outcome]) + lines.extend( + [ + "", + f"Provenance: repo={repo.identity}; branch={repo.branch}; head={repo.head_sha}; packet={packet_id}", + ] + ) + content = "\n".join(lines) + return bounded(content, MAX_EPISODE_CHARS), structured + + +def build_semantic_evidence(structured: dict[str, Any]) -> str: + """Format changed paths for memory extraction. + + Test and build results remain in the local evidence store for diagnostics, + but are not useful repository knowledge by default and should not steer + memory extraction toward transient verification details. + """ + modified_paths = [ + bounded(path, 500) for path in structured.get("files_modified", [])[:20] + ] + commands = structured.get("commands") or [] + if not any(command.get("status") == "failed" for command in commands): + commands = [] + + if not modified_paths and not commands: + return "" + + lines = ["Additional repository details from this session"] + if modified_paths: + lines.extend( + [ + "", + "Changed paths:", + *[f"- {path}" for path in modified_paths], + ] + ) + if commands: + lines.extend(["", "Commands run in this session:", *_render_command_lines(commands)]) + return bounded("\n".join(lines), 8000) + + +def build_extraction_messages(structured: dict[str, Any]) -> list[dict[str, str]]: + """Build the session messages sent to Mem0 for memory extraction.""" + evidence = build_semantic_evidence(structured) + messages = [ + {"role": message["role"], "content": redact(message["content"]).strip()} + for message in structured.get("extraction_messages", []) + if message.get("role") in {"user", "assistant"} and message.get("content") + ] + if evidence: + for message in reversed(messages): + if message["role"] == "assistant": + message["content"] = f"{message['content']}\n\n{evidence}" + break + else: + messages.append({"role": "assistant", "content": evidence}) + + return messages + + +def _estimated_tokens(value: str) -> int: + """Conservatively estimate tokens without adding a tokenizer dependency.""" + ascii_chars = sum(ord(char) < 128 for char in value) + return math.ceil((ascii_chars * 0.4) + (len(value) - ascii_chars)) + + +def _message_tokens(messages: list[dict[str, str]]) -> int: + return _estimated_tokens(json.dumps(messages, ensure_ascii=False)) + + +def _is_agent_assignment(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent assignment (") + + +def _is_agent_response(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent response (") + + +def extraction_message_batches( + messages: list[dict[str, str]], + *, + max_tokens: int = MAX_EXTRACTION_INPUT_TOKENS, +) -> list[list[dict[str, str]]]: + """Keep exchanges together when possible; split oversized messages to enforce the request budget.""" + if not messages or _message_tokens(messages) <= max_tokens: + return [messages] + + exchanges: list[list[dict[str, str]]] = [] + exchange: list[dict[str, str]] = [] + for message in messages: + if message.get("role") == "user" and exchange: + exchanges.append(exchange) + exchange = [] + exchange.append(message) + if exchange: + exchanges.append(exchange) + + units: list[list[dict[str, str]]] = [] + for exchange in exchanges: + if _message_tokens(exchange) <= max_tokens: + units.append(exchange) + continue + index = 0 + while index < len(exchange): + message = exchange[index] + if ( + _is_agent_assignment(message) + and index + 1 < len(exchange) + and _is_agent_response(exchange[index + 1]) + ): + units.append(exchange[index : index + 2]) + index += 2 + else: + units.append([message]) + index += 1 + + bounded_units: list[list[dict[str, str]]] = [] + for unit in units: + if _message_tokens(unit) <= max_tokens: + bounded_units.append(unit) + continue + for message in unit: + remaining = message["content"] + while remaining: + low, high = 0, len(remaining) + while low < high: + middle = (low + high + 1) // 2 + if _message_tokens([{**message, "content": remaining[:middle]}]) <= max_tokens: + low = middle + else: + high = middle - 1 + if low == 0: + raise ValueError("Extraction token budget cannot fit a message") + bounded_units.append([{**message, "content": remaining[:low]}]) + remaining = remaining[low:] + + batches: list[list[dict[str, str]]] = [] + batch: list[dict[str, str]] = [] + for unit in bounded_units: + candidate = [*batch, *unit] + if batch and _message_tokens(candidate) > max_tokens: + batches.append(batch) + batch = list(unit) + else: + batch = candidate + if batch: + batches.append(batch) + return batches + + +def _request_json( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + raw = json.dumps(payload, ensure_ascii=False).encode() + request = urllib.request.Request( + url, + data=raw, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + parsed = json.loads(response_raw or b"{}") + return parsed, len(raw), len(response_raw) + + +def _request_json_with_network_retry( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + """Retry one transient connection failure without retrying API responses.""" + try: + return _request_json(url, key, payload, timeout) + except urllib.error.HTTPError: + raise + except (urllib.error.URLError, TimeoutError, OSError): + time.sleep(0.25) + return _request_json(url, key, payload, timeout) + + +def _get_json( + url: str, key: str, timeout: float +) -> tuple[dict[str, Any] | list[Any], int]: + request = urllib.request.Request( + url, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="GET", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + return json.loads(response_raw or b"{}"), len(response_raw) + + +def _event_id(response: dict[str, Any] | list[Any]) -> str: + return str(response.get("event_id", "")) if isinstance(response, dict) else "" + + +def _stored_event_ids(value: Any) -> list[str]: + raw = str(value or "") + if not raw.startswith("["): + return [] + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return [] + return [str(item or "") for item in parsed] if isinstance(parsed, list) else [] + + +def _result_count(response: dict[str, Any] | list[Any]) -> int: + if isinstance(response, dict): + results = response.get("results") + return len(results) if isinstance(results, list) else 0 + return len(response) if isinstance(response, list) else 0 + + +def touch_handoff_heartbeat() -> None: + """Mark the worker's handoff file alive so recovery does not relaunch it.""" + path = os.environ.get("MEM0_CODE_HANDOFF_PATH", "") + if not path: + return + try: + os.utime(path) + except OSError: + pass + + +def _wait_for_event(api_url: str, key: str, event_id: str) -> tuple[str, int, int]: + """Wait for extraction to finish before a later task can search the store.""" + if not event_id: + return "MISSING", 0, 0 + wait_seconds = float(os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120")) + poll_seconds = max(float(os.environ.get("MEM0_CODE_EVENT_POLL_SECONDS", "1")), 0.1) + deadline = time.monotonic() + wait_seconds + response_chars = 0 + while time.monotonic() < deadline: + touch_handoff_heartbeat() + try: + response, size = _get_json( + f"{api_url}/v1/event/{event_id}/", + key, + min(10, poll_seconds + 5), + ) + except urllib.error.HTTPError as exc: + if exc.code not in {408, 429} and exc.code < 500: + raise + time.sleep(poll_seconds) + continue + except (urllib.error.URLError, TimeoutError, OSError): + # The extraction job is durable server-side. A transient polling + # failure must not discard a job that may still complete normally. + time.sleep(poll_seconds) + continue + response_chars += size + status = ( + str(response.get("status", "UNKNOWN")) + if isinstance(response, dict) + else "UNKNOWN" + ) + if status in {"SUCCEEDED", "FAILED"}: + return status, response_chars, _result_count(response) + time.sleep(poll_seconds) + return "TIMEOUT", response_chars, 0 + + +def _record_flush( + repo: RepoContext, + session_id: str, + reason: str, + status: str, + elapsed: float, + **extra: Any, +) -> None: + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status=status, + success=status in {"semantic-succeeded", "nothing-to-flush"}, + duration_ms=round(elapsed, 2), + **extra, + ) + + +def flush_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + key = api_key() + if not key: + telemetry.record("flush", reason=reason, status="local-only", success=False) + return {"status": "local-only", "reason": "no-api-key"} + + session_id = _session_id(hook_input) + if session_id == "unknown-session": + telemetry.record("flush", reason=reason, status="no-session-id", success=False) + return {"status": "error", "reason": "no-session-id"} + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prepared = store.prepare_flush(repo, session_id, reason) + if prepared is None: + return {"status": "nothing-to-flush"} + packet_id, events = prepared + existing_flush = store.flush_record(packet_id) or {} + + _, structured = build_episode( + repo, + session_id, + packet_id, + events, + canonical_task=bounded(hook_input.get("task", ""), 4000), + task_outcome=bounded(hook_input.get("task_outcome", ""), 2000), + ) + + metadata = {"source": _harness_source_tag} + if repo.branch and repo.branch not in {"detached", "unknown"}: + metadata["branch"] = repo.branch + if repo.head_sha: + metadata["git_sha"] = repo.head_sha + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + add_url = f"{api_url}/v3/memories/add/" + + write_user = _scope_value(user_id()) + write_project = _scope_value(repo.project_id) + if not write_user or not _scope_value(repo.app_id) or not write_project: + telemetry.record("flush", reason=reason, status="unscoped", success=False) + return {"status": "error", "reason": "wildcard-scope"} + + body = { + "agent_id": write_project, + "user_id": write_user, + "app_id": repo.app_id, + "run_id": session_id, + "metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)}, + "agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS, + "custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS, + "custom_categories": CODING_MEMORY_CATEGORIES, + "infer": True, + } + + started = time.perf_counter() + try: + stored_events = _stored_event_ids(existing_flush.get("semantic_event_id")) + existing_event = ( + "" if stored_events else str(existing_flush.get("semantic_event_id") or "") + ) + if existing_event: + existing_status, existing_resp, existing_items = _wait_for_event( + api_url, key, existing_event + ) + if existing_status == "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=existing_event, + error="", + ) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + True, + item_count=existing_items, + response_chars=existing_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=existing_items, + resumed=True, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "semantic_status": existing_status, + "memory_count": existing_items, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + if existing_status == "TIMEOUT": + elapsed = (time.perf_counter() - started) * 1000 + error = "semantic extraction event timed out" + store.update_flush(packet_id, status="semantic-timeout", error=error) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + False, + response_chars=existing_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-timeout", + elapsed, + resumed=True, + error_kind="timeout", + ) + return { + "status": "semantic-timeout", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + + message_batches = [ + batch + for batch in extraction_message_batches( + build_extraction_messages(structured) + ) + if batch + ] + batches = [(body, messages) for messages in message_batches] + if not batches: + store.update_flush(packet_id, status="semantic-succeeded", error="") + return {"status": "nothing-to-flush", "packet_id": packet_id} + operation_name = "flush-retry" if stored_events else "flush" + semantic_events = stored_events[: len(batches)] + semantic_events += [""] * (len(batches) - len(semantic_events)) + semantic_req = 0 + semantic_resp = 0 + for index, (body, messages) in enumerate(batches): + if semantic_events[index]: + continue + semantic_response, request_chars, response_chars = _request_json( + add_url, + key, + {**body, "messages": messages}, + 15, + ) + semantic_events[index] = _event_id(semantic_response) + semantic_req += request_chars + semantic_resp += response_chars + store.update_flush( + packet_id, + status="semantic-queued", + semantic_event_id=json.dumps(semantic_events), + ) + + semantic_event = semantic_events[-1] + semantic_status = "SUCCEEDED" + event_resp = 0 + semantic_items = 0 + failed_event = semantic_event + for index, queued_event in enumerate(semantic_events): + status, response_chars, item_count = _wait_for_event( + api_url, key, queued_event + ) + event_resp += response_chars + semantic_items += item_count + if status != "SUCCEEDED": + semantic_status = status + failed_event = queued_event + if status in {"FAILED", "MISSING"}: + semantic_events[index] = "" + store.update_flush( + packet_id, semantic_event_id=json.dumps(semantic_events) + ) + break + if semantic_status != "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + error = f"semantic extraction event {semantic_status.lower()}" + store.update_flush( + packet_id, status=f"semantic-{semantic_status.lower()}", error=error + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + False, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + f"semantic-{semantic_status.lower()}", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + error_kind=telemetry.error_kind(error), + ) + return { + "status": f"semantic-{semantic_status.lower()}", + "packet_id": packet_id, + "semantic_event_id": failed_event, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=json.dumps(semantic_events), + error="", + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + True, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + request_chars=semantic_req, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": semantic_event, + "semantic_status": semantic_status, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + except Exception as exc: # hooks must fail open + elapsed = (time.perf_counter() - started) * 1000 + error = bounded(str(exc), 1000) + store.update_flush(packet_id, status="error", error=error) + store.operation(repo, session_id, "flush", elapsed, False, error=error) + _record_flush( + repo, + session_id, + reason, + "error", + elapsed, + error_kind=telemetry.error_kind(exc), + ) + return {"status": "error", "packet_id": packet_id, "error": error} + + +def checkpoint_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + """Run remote extraction at a durable boundary.""" + return flush_session(store, hook_input, reason) + + + +def search_memories( + store: EvidenceStore | None, + repo: RepoContext, + session_id: str | None, + query: str, + *, + top_k: int | None = None, + category: str | None = None, + scope: str | None = None, + run_id: str | None = None, + operation: str = "search", + timeout: float = 5, +) -> MemorySearchResult: + key = api_key() + if not key or not query.strip(): + return MemorySearchResult(False, 0, 0, []) + search_once = os.environ.get( + "MEM0_CODE_SEARCH_ONCE_PER_SESSION", "false" + ).lower() in { + "1", + "true", + "yes", + "on", + } + track_session = store is not None and bool(session_id) + if ( + search_once + and track_session + and store.has_operation(repo.identity, session_id, "search") + ): + return MemorySearchResult(False, 0, 0, []) + + result_limit = min( + max( + top_k + if top_k is not None + else _int_option("top_k", "MEM0_CODE_TOP_K", 3), + 1, + ), + 20, + ) + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ValueError(f"Unknown memory category: {category}") + user, project = _scope_value(user_id()), _scope_value(repo.project_id) + if not user or not project or not _scope_value(repo.app_id): + return MemorySearchResult(False, 0, 0, []) + filters = _search_filters(user, repo, resolve_search_scope(scope)) + if category: + filters = {"AND": [filters, {"categories": {"contains": category}}]} + if run_id: + filters = {"AND": [filters, {"run_id": run_id}]} + payload = { + "query": query, + "app_id": repo.app_id, + "filters": filters, + "top_k": result_limit, + "rerank": False, + "latest_only": True, + } + url = ( + os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + + "/v3/memories/search/" + ) + started = time.perf_counter() + try: + response, request_chars, response_chars = _request_json_with_network_retry( + url, key, payload, timeout + ) + memories = ( + response if isinstance(response, list) else response.get("results", []) + ) + memories = [ + memory + for memory in memories + if isinstance(memory, dict) + and (memory.get("metadata") or {}).get("record_kind") != "task_episode" + ][:result_limit] + if track_session: + returned_memories = store.unseen(session_id, repo.identity, memories) + store.mark_injected(session_id, repo.identity, returned_memories) + already_shown_count = len(memories) - len(returned_memories) + else: + returned_memories = memories + already_shown_count = 0 + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation( + repo, + session_id, + operation, + elapsed, + True, + item_count=len(returned_memories), + request_chars=request_chars, + response_chars=response_chars, + ) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=True, + duration_ms=round(elapsed, 2), + matched_count=len(memories), + returned_count=len(returned_memories), + already_shown_count=already_shown_count, + top_k=result_limit, + has_category=bool(category), + ) + return MemorySearchResult( + succeeded=True, + matched_count=len(memories), + already_shown_count=already_shown_count, + memories=returned_memories, + ) + except Exception as exc: + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation(repo, session_id, operation, elapsed, False, error=str(exc)) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=False, + duration_ms=round(elapsed, 2), + top_k=result_limit, + has_category=bool(category), + error_kind=telemetry.error_kind(exc), + ) + return MemorySearchResult(False, 0, 0, []) + + +def format_context( + memories: list[dict[str, Any]], + heading: str = "Relevant repository memories:", +) -> str: + if not memories: + return "" + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + lines = [heading] if heading else [] + for memory in memories: + text = re.sub( + r"\s+", + " ", + redact(memory.get("memory") or memory.get("text") or ""), + ) + text = text.strip() + if not text: + continue + branch = str((memory.get("metadata") or {}).get("branch") or "").strip() + branch_label = ( + f" [learnt on branch {branch}]" + if branch.casefold() not in {"", "main", "master", "unknown", "detached"} + else "" + ) + number = len(lines) if heading else len(lines) + 1 + entry = f"{number}. {text}{branch_label}" + candidate = "\n".join([*lines, entry]) + if len(candidate) <= limit: + lines.append(entry) + continue + if not lines or (heading and len(lines) == 1): + prefix = f"{number}. " + suffix = f"…{branch_label}" + available = ( + limit + - len("\n".join(lines)) + - (1 if lines else 0) + - len(prefix) + - len(suffix) + ) + if available > 0: + lines.append(prefix + text[:available].rstrip() + suffix) + break + minimum_lines = 2 if heading else 1 + return "\n".join(lines) if len(lines) >= minimum_lines else "" + + +def format_search_result(result: MemorySearchResult) -> str: + """Return only the text the coding agent needs from an explicit memory search.""" + if not result.succeeded: + return "Memory search failed." + if result.memories: + rendered = format_context(result.memories, heading="") + if rendered: + return rendered + return "No matching memories found." + + +def combine_context(*contexts: str) -> str: + """Combine memory sources under one hard budget without repeated lines.""" + seen: set[str] = set() + lines: list[str] = [] + for context in contexts: + for line in str(context or "").splitlines(): + normalized = re.sub(r"\s+", " ", line).strip().casefold() + if not normalized or normalized in seen: + continue + seen.add(normalized) + lines.append(line.rstrip()) + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + return bounded("\n".join(lines), limit) if lines else "" + + +def _scoped_memory_ids( + api_url: str, key: str, user: str, repo: RepoContext, include_project: bool +) -> list[str]: + """List this user's memory ids for this repository, plus the shared project memory when asked.""" + ids: list[str] = [] + seen: set[str] = set() + prefix = repo.app_id + _collect_memory_ids( + api_url, key, {"user_id": user}, ids, seen, + app_id_prefix=prefix, + ) + if include_project: + for project_id in _shared_project_ids(repo): + _collect_memory_ids( + api_url, key, {"agent_id": project_id}, ids, seen, + app_id_prefix=prefix, + ) + return ids + + +def _collect_memory_ids( + api_url: str, + key: str, + filters: dict[str, Any], + ids: list[str], + seen: set[str], + *, + app_id_prefix: str = "", +) -> None: + """Page through one list filter; the list endpoint returns nothing for an OR whose user branch has no memories.""" + payload = {"filters": filters} + for page in range(1, FORGET_MAX_PAGES + 1): + parsed, _, _ = _request_json( + f"{api_url}/v2/memories/?page={page}&page_size={FORGET_PAGE_SIZE}", + key, + payload, + 15, + ) + items = parsed.get("results") if isinstance(parsed, dict) else parsed + if not isinstance(items, list) or not items: + break + for item in items: + if not isinstance(item, dict): + continue + memory_id = str(item.get("id", "")) + if not memory_id or memory_id in seen: + continue + if app_id_prefix: + item_app_id = str(item.get("app_id") or "") + if item_app_id != app_id_prefix and not item_app_id.startswith(app_id_prefix + "/"): + continue + seen.add(memory_id) + ids.append(memory_id) + if len(items) < FORGET_PAGE_SIZE: + break + + +def _delete_memory(api_url: str, key: str, memory_id: str) -> bool: + request = urllib.request.Request( + f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/", + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="DELETE", + ) + try: + with urllib.request.urlopen(request, timeout=15): + return True + except Exception: + return False + + +def forget_remote_repo( + repo: RepoContext, *, include_project_memory: bool = False +) -> dict[str, Any]: + """Delete this user's memories for this repository; project memory is shared, so only on request.""" + key = api_key() + if not key: + telemetry.record("forget", repo=repo, success=False, error_kind="no-api-key") + return {"status": "error", "error": "Mem0 API key is not configured"} + user = _scope_value(user_id()) + if not user or not _scope_value(repo.app_id) or not _scope_value(repo.project_id): + telemetry.record("forget", repo=repo, success=False, error_kind="unscoped") + return { + "status": "error", + "error": "Refusing to forget: the user or repository scope is a wildcard", + } + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + try: + memory_ids = _scoped_memory_ids(api_url, key, user, repo, include_project_memory) + except Exception as exc: + telemetry.record( + "forget", repo=repo, success=False, error_kind=telemetry.error_kind(exc) + ) + return {"status": "error", "error": bounded(str(exc), 1000)} + deleted = sum(_delete_memory(api_url, key, memory_id) for memory_id in memory_ids) + failed = len(memory_ids) - deleted + telemetry.record("forget", repo=repo, success=not failed, item_count=deleted) + if failed: + return { + "status": "partial", + "deleted": deleted, + "failed": failed, + "error": f"{failed} of {len(memory_ids)} memories could not be deleted", + } + return {"status": "deleted", "deleted": deleted} + + +def _doctor_mem0_authentication(repo: RepoContext) -> dict[str, Any]: + """Verify the configured key with one read-only, repository-scoped search.""" + key = api_key() + if not key: + return {"ok": False, "detail": "API key missing"} + payload = { + "query": "Mem0 authentication check", + "filters": { + "AND": [ + {"user_id": user_id()}, + {"app_id": repo.app_id}, + ] + }, + "top_k": 1, + "threshold": 1.0, + "rerank": False, + } + url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + started = time.perf_counter() + try: + _request_json(f"{url}/v3/memories/search/", key, payload, 5) + except Exception as exc: + return {"ok": False, "detail": bounded(str(exc), 300)} + elapsed = (time.perf_counter() - started) * 1000 + return {"ok": True, "detail": f"connected ({elapsed:.0f} ms)"} + + +def _doctor_user_id() -> dict[str, Any]: + """Flag a configured user ID the plugin refuses, since the silent fallback surprises people.""" + configured = _plugin_option("user_id", "MEM0_CODE_USER_ID") or os.environ.get( + "MEM0_USER_ID", "" + ) + if configured and not _scope_value(configured): + return { + "ok": False, + "detail": f"configured user_id {configured!r} is a wildcard; using {user_id()!r}", + } + return {"ok": True, "detail": user_id()} + + +def doctor(cwd: str | None = None) -> dict[str, Any]: + repo = resolve_repo(cwd) + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + checks: dict[str, dict[str, Any]] = { + "python": { + "ok": tuple(sys.version_info[:2]) >= (3, 10), + "detail": f"{sys.version_info.major}.{sys.version_info.minor}", + }, + "data_directory": { + "ok": os.access(directory, os.W_OK), + "detail": str(directory), + }, + "mem0_api_key": { + "ok": bool(api_key()), + "detail": "configured" if api_key() else "missing", + }, + "repository": { + "ok": bool(repo.identity), + "detail": repo.identity, + }, + "user_id": _doctor_user_id(), + "mem0_authentication": _doctor_mem0_authentication(repo), + } + return { + "ok": all(bool(value["ok"]) for value in checks.values()), + "plugin_version": PLUGIN_VERSION, + "repo_id": repo.identity, + "app_id": repo.app_id, + "user_id": user_id(), + "checks": checks, + } diff --git a/integrations/kimi-plugin/core/telemetry.py b/integrations/kimi-plugin/core/telemetry.py new file mode 100644 index 000000000..249595475 --- /dev/null +++ b/integrations/kimi-plugin/core/telemetry.py @@ -0,0 +1,397 @@ +#!/usr/bin/env python3 +"""Anonymous usage telemetry for Mem0 agent plugins. + +Hooks run on a 3-6 second budget and fire on every tool call, so recording never +touches the network: `record` appends one JSON line to a local spool and returns. +A detached `python3 telemetry.py` drains the spool in one batched PostHog request, +started once per session and again from the flush worker that is already detached. + +Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false. + +Never sends prompts, memory text, queries, file paths, repository names, or API +keys: only event names, durations, counts, coarse outcomes, and salted hashes. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import platform +import subprocess +import sys +import time +import urllib.error +import urllib.request +import uuid +from pathlib import Path +from typing import Any + +import memory_core + +_harness: str = "generic" +_source_tag: str = "MEM0_PLUGIN" +_PRIVATE_KEYS = { + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +} + + +def init(harness: str = "generic", source_tag: str = "") -> None: + global _harness, _source_tag + _harness = harness + _source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN" + +POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" +POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" +POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/" +EVENT_PREFIX = "code" +SPOOL_LIMIT_BYTES = 256 * 1024 +BATCH_SIZE = 100 +SEND_TIMEOUT = 5 +CLAIM_STALE_SECONDS = 120 +CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60 + + +def is_enabled() -> bool: + """Whether telemetry is switched on for this process.""" + return os.environ.get("MEM0_TELEMETRY", "true").strip().lower() not in { + "false", + "0", + "no", + "off", + } + + +def _digest(value: str, length: int = 16) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length] + + +def _safe_value(value: Any) -> Any: + if isinstance(value, str): + return memory_core.redact(value) + if isinstance(value, dict): + return { + key: _safe_value(item) + for key, item in value.items() + if "".join(character for character in str(key).lower() if character.isalnum()) + not in _PRIVATE_KEYS + } + if isinstance(value, (list, tuple)): + return [_safe_value(item) for item in value] + if value is None or isinstance(value, (bool, int, float)): + return value + return memory_core.redact(value) + + +def _spool_path() -> Path: + return memory_core.data_dir() / "telemetry.jsonl" + + +def _identity_path() -> Path: + return memory_core.data_dir() / "telemetry-identity.json" + + +def _read_identity() -> dict[str, str]: + try: + value = json.loads(_identity_path().read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + return value if isinstance(value, dict) else {} + + +def _write_identity(identity: dict[str, str]) -> None: + path = _identity_path() + temporary = path.with_suffix(f".{os.getpid()}.tmp") + try: + path.parent.mkdir(parents=True, exist_ok=True) + temporary.write_text(json.dumps(identity), encoding="utf-8") + temporary.replace(path) + except OSError: + try: + temporary.unlink() + except OSError: + pass + + +def anonymous_id(identity: dict[str, str] | None = None) -> str: + """Per-machine anonymous identifier, created and persisted on first use.""" + identity = _read_identity() if identity is None else identity + existing = identity.get("anonymous_id") + if existing: + return existing + created = f"code-anon-{uuid.uuid4().hex}" + identity["anonymous_id"] = created + _write_identity(identity) + return created + + +def is_first_run() -> bool: + """Whether this machine has never recorded a plugin event before.""" + return not _identity_path().exists() + + +def record( + event: str, + *, + repo: Any = None, + session_id: str | None = None, + **properties: Any, +) -> None: + """Append one event to the local spool. Never blocks and never raises.""" + if not is_enabled(): + return + try: + spool = _spool_path() + try: + if spool.stat().st_size > SPOOL_LIMIT_BYTES: + return + except OSError: + pass + properties = _safe_value(properties) + properties.update( + harness=_harness, + plugin_version=memory_core.PLUGIN_VERSION, + os=sys.platform, + python_version=platform.python_version(), + ) + if repo is not None: + properties["repo_hash"] = _digest(getattr(repo, "identity", "")) + if session_id: + properties["session_hash"] = _digest(session_id) + line = json.dumps( + { + "event": f"{EVENT_PREFIX}.{event}", + "timestamp": memory_core.utc_now(), + "properties": { + key: value for key, value in properties.items() if value is not None + }, + }, + separators=(",", ":"), + default=str, + ) + spool.parent.mkdir(parents=True, exist_ok=True) + with spool.open("a", encoding="utf-8") as handle: + handle.write(line + "\n") + except Exception: + pass + + +def error_kind(exc: BaseException | str) -> str: + """Coarse, content-free label for a failure, safe to send.""" + text = exc if isinstance(exc, str) else f"{type(exc).__name__}: {exc}" + lowered = text.lower() + if "timed out" in lowered or "timeout" in lowered: + return "timeout" + if "401" in lowered or "403" in lowered or "unauthor" in lowered or "forbidden" in lowered: + return "auth" + if "429" in lowered or "rate limit" in lowered: + return "rate-limited" + if any(code in lowered for code in ("500", "502", "503", "504")): + return "server-error" + if "400" in lowered or "422" in lowered: + return "bad-request" + if isinstance(exc, str): + return "other" + if isinstance(exc, urllib.error.URLError): + return "network" + return type(exc).__name__ + + +def spawn_flush() -> bool: + """Start the detached sender that drains the spool.""" + if not is_enabled(): + return False + try: + if not _spool_path().exists() and not any( + memory_core.data_dir().glob("telemetry-*.sending") + ): + return False + subprocess.Popen( + [sys.executable, str(Path(__file__).resolve())], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + close_fds=True, + **memory_core.detached_process_kwargs(), + ) + return True + except Exception: + return False + + +def _claim_spool() -> Path | None: + """Rename the spool aside so exactly one sender owns each batch.""" + directory = memory_core.data_dir() + claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending" + spool = _spool_path() + try: + spool.replace(claim) + return claim + except OSError: + pass + now = time.time() + for orphan in sorted(directory.glob("telemetry-*.sending")): + try: + age = now - orphan.stat().st_mtime + except OSError: + continue + if age > CLAIM_EXPIRY_SECONDS: + try: + orphan.unlink() + except OSError: + pass + continue + if age < CLAIM_STALE_SECONDS: + continue + try: + orphan.replace(claim) + return claim + except OSError: + continue + return None + + +def _resolve_email(key: str) -> str: + """Trade the API key for the account email so events join other Mem0 surfaces.""" + url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/" + request = urllib.request.Request( + url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json"} + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT) as response: + payload = json.loads(response.read().decode("utf-8")) + except Exception: + return "" + email = payload.get("user_email") if isinstance(payload, dict) else "" + return email if isinstance(email, str) else "" + + +def _post(payload: dict[str, Any], url: str) -> bool: + request = urllib.request.Request( + url, + data=json.dumps(payload, default=str).encode("utf-8"), + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT): + return True + except Exception: + return False + + +def resolve_distinct_id() -> tuple[str, str]: + """Return the PostHog distinct id and the anonymous id it replaced, if any.""" + identity = _read_identity() + email = identity.get("email", "") + if email: + return email, "" + key = memory_core.api_key() + if not key: + return anonymous_id(identity), "" + email = _resolve_email(key) + if not email: + return anonymous_id(identity), "" + previous = identity.get("anonymous_id", "") + identity["email"] = email + _write_identity(identity) + return email, previous + + +def flush() -> int: + """Drain claimed spools to PostHog and return the number of events sent.""" + if not is_enabled(): + return 0 + claim = _claim_spool() + if claim is None: + return 0 + try: + lines = claim.read_text(encoding="utf-8").splitlines() + except OSError: + return 0 + events = [] + for line in lines: + try: + value = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(value, dict) and value.get("event"): + events.append(value) + if not events: + try: + claim.unlink() + except OSError: + pass + return 0 + + distinct_id, aliased_anonymous_id = resolve_distinct_id() + if aliased_anonymous_id: + _post( + { + "api_key": POSTHOG_API_KEY, + "event": "$identify", + "distinct_id": distinct_id, + "properties": { + "$anon_distinct_id": aliased_anonymous_id, + "$lib": "posthog-python", + }, + }, + POSTHOG_CAPTURE_URL, + ) + + sent = 0 + for start in range(0, len(events), BATCH_SIZE): + batch = [ + { + "event": event["event"], + "distinct_id": distinct_id, + "timestamp": event.get("timestamp"), + "properties": { + "source": _source_tag, + "language": "python", + "$process_person_profile": False, + "$lib": "posthog-python", + **(event.get("properties") or {}), + }, + } + for event in events[start : start + BATCH_SIZE] + ] + if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL): + return sent + sent += len(batch) + try: + claim.unlink() + except OSError: + pass + return sent + + +def main() -> int: + flush() + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception: + raise SystemExit(0) diff --git a/integrations/kimi-plugin/hooks/adapter.py b/integrations/kimi-plugin/hooks/adapter.py new file mode 100644 index 000000000..3a16b52d3 --- /dev/null +++ b/integrations/kimi-plugin/hooks/adapter.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Translate Kimi Code hooks into the shared Mem0 runtime.""" + +from __future__ import annotations + +import contextlib +import io +import json +import os +import sys +import uuid +from pathlib import Path + +HERE = Path(__file__).resolve() +BUNDLED_CORE = HERE.parent.parent / "core" +CORE = BUNDLED_CORE if BUNDLED_CORE.is_dir() else HERE.parents[2] / "core" / "python" +sys.path.insert(0, str(CORE)) + +import hook_runner # noqa: E402 +import telemetry # noqa: E402 +from memory_core import configure_harness, record_sidekick_start, record_sidekick_stop, record_tool # noqa: E402 + +EVENTS = { + "SessionStart": ["session-start"], + "UserPromptSubmit": ["user-prompt"], + "PostToolUse": ["post-tool"], + "PostToolUseFailure": ["post-tool-failure"], + "Stop": ["stop"], + "PreCompact": ["flush", "--reason", "pre-compact"], + "SessionEnd": ["flush", "--reason", "session-end"], + "SubagentStart": ["sidekick-start"], + "SubagentStop": ["sidekick-stop"], +} + + +def _session_dir(session_id: str) -> Path | None: + home = Path(os.environ.get("KIMI_CODE_HOME", Path.home() / ".kimi-code")).expanduser() + found = None + try: + with (home / "session_index.jsonl").open(encoding="utf-8") as index: + for line in index: + try: + entry = json.loads(line) + except json.JSONDecodeError: + continue + if entry.get("sessionId") == session_id and isinstance(entry.get("sessionDir"), str): + found = Path(entry["sessionDir"]) + except OSError: + pass + if found: + return found + if not session_id or Path(session_id).name != session_id: + return None + return next((path for path in (home / "sessions").glob(f"*/{session_id}") if path.is_dir()), None) + + +def _last_assistant_message(transcript: Path) -> str: + message = "" + step_id = None + parts = [] + try: + lines = transcript.read_text(encoding="utf-8").splitlines() + except OSError: + return "" + for line in lines: + try: + record = json.loads(line) + except json.JSONDecodeError: + continue + if record.get("type") != "context.append_loop_event" or not isinstance(record.get("event"), dict): + continue + event = record["event"] + if event.get("type") == "step.begin": + step_id = event.get("uuid") + parts = [] + elif event.get("type") == "content.part" and event.get("stepUuid") == step_id: + part = event.get("part") + if isinstance(part, dict) and part.get("type") == "text" and isinstance(part.get("text"), str): + parts.append(part["text"]) + elif event.get("type") == "step.end" and event.get("uuid") == step_id: + if event.get("finishReason") not in {"error", "interrupted"} and parts: + message = "".join(parts) + step_id = None + parts = [] + return message + + +def normalize(payload: dict) -> dict: + value = dict(payload) + if "tool_output" in value: + value.setdefault("tool_response", value["tool_output"]) + if "error" in value: + value.setdefault("tool_response", value["error"]) + if "response" in value: + value.setdefault("last_assistant_message", value["response"]) + if "agent_name" in value: + value.setdefault("agent_type", value["agent_name"]) + if value.get("hook_event_name") == "Stop": + session = _session_dir(str(value.get("session_id") or "")) + transcript = session / "agents" / "main" / "wire.jsonl" if session else None + if transcript and transcript.is_file(): + value.setdefault("transcript_path", str(transcript)) + if message := _last_assistant_message(transcript): + value.setdefault("last_assistant_message", message) + return value + + +def _sidekick_start(store, payload): + payload = dict(payload) + payload.setdefault("agent_id", f"kimi-{uuid.uuid4().hex}") + context = record_sidekick_start(store, payload) + return {"hookSpecificOutput": {"additionalContext": context}} if context else None + + +def _sidekick_stop(store, payload): + record_sidekick_stop(store, payload) + + +def main() -> int: + if len(sys.argv) != 2 or sys.argv[1] not in EVENTS: + return 2 + event = sys.argv[1] + try: + raw = json.load(sys.stdin) + except (json.JSONDecodeError, OSError): + raw = {} + sys.argv = [sys.argv[0], *EVENTS[event]] + sys.stdin = io.StringIO(json.dumps(normalize(raw if isinstance(raw, dict) else {}))) + configure_harness("kimi", data_dir_name="kimi-plugin", source_tag="kimi_plugin") + telemetry.init(harness="kimi", source_tag="KIMI_PLUGIN") + output = io.StringIO() + with contextlib.redirect_stdout(output): + result = hook_runner.run( + extra_actions={ + "post-tool-failure": lambda store, payload: record_tool(store, payload, failed=True), + "sidekick-start": _sidekick_start, + "sidekick-stop": _sidekick_stop, + }, + automatic_flush_reasons={"session-end", "pre-compact"}, + ) + if event in {"UserPromptSubmit", "SubagentStart"} and (raw_output := output.getvalue().strip()): + parsed = json.loads(raw_output) + context = parsed.get("hookSpecificOutput", {}).get("additionalContext", "") + if context: + print(context) + return result + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception as exc: + hook_runner.log_failure(exc) + raise SystemExit(0) from None diff --git a/integrations/kimi-plugin/kimi.plugin.json b/integrations/kimi-plugin/kimi.plugin.json new file mode 100644 index 000000000..062a783ee --- /dev/null +++ b/integrations/kimi-plugin/kimi.plugin.json @@ -0,0 +1,32 @@ +{ + "name": "mem0", + "version": "0.3.1", + "description": "Cross-session memory and token savings for coding agents.", + "keywords": ["memory", "coding-agents", "continual-learning", "token-efficiency"], + "author": { "name": "Mem0", "email": "support@mem0.ai" }, + "homepage": "https://docs.mem0.ai/integrations/kimi", + "license": "Apache-2.0", + "interface": { + "displayName": "Mem0", + "shortDescription": "Cross-session memory for Kimi Code" + }, + "skills": "./skills/", + "agents": "./agents/", + "mcpServers": { + "mem0": { + "command": "python3", + "args": ["./core/mcp_server.py"] + } + }, + "hooks": [ + { "event": "SessionStart", "command": "python3 ./hooks/adapter.py SessionStart", "timeout": 5 }, + { "event": "UserPromptSubmit", "command": "python3 ./hooks/adapter.py UserPromptSubmit", "timeout": 6 }, + { "event": "PostToolUse", "matcher": ".*", "command": "python3 ./hooks/adapter.py PostToolUse", "timeout": 3 }, + { "event": "PostToolUseFailure", "matcher": ".*", "command": "python3 ./hooks/adapter.py PostToolUseFailure", "timeout": 3 }, + { "event": "Stop", "command": "python3 ./hooks/adapter.py Stop", "timeout": 3 }, + { "event": "PreCompact", "command": "python3 ./hooks/adapter.py PreCompact", "timeout": 5 }, + { "event": "SessionEnd", "command": "python3 ./hooks/adapter.py SessionEnd", "timeout": 5 }, + { "event": "SubagentStart", "matcher": "^sidekick$", "command": "python3 ./hooks/adapter.py SubagentStart", "timeout": 5 }, + { "event": "SubagentStop", "matcher": "^sidekick$", "command": "python3 ./hooks/adapter.py SubagentStop", "timeout": 5 } + ] +} diff --git a/integrations/kimi-plugin/plugin-build.json b/integrations/kimi-plugin/plugin-build.json new file mode 100644 index 000000000..021a03a80 --- /dev/null +++ b/integrations/kimi-plugin/plugin-build.json @@ -0,0 +1,13 @@ +{ + "id": "mem0", + "version": "0.3.1", + "homepage": "https://docs.mem0.ai/integrations/kimi", + "native": { + "pluginRoot": "${KIMI_PLUGIN_ROOT}", + "files": { + "kimi.plugin.json": "kimi.plugin.json", + "hooks/adapter.py": "hooks/adapter.py", + "agents/sidekick.md": "agents/sidekick.md" + } + } +} diff --git a/integrations/kimi-plugin/skills/forget/SKILL.md b/integrations/kimi-plugin/skills/forget/SKILL.md new file mode 100644 index 000000000..f4709793e --- /dev/null +++ b/integrations/kimi-plugin/skills/forget/SKILL.md @@ -0,0 +1,26 @@ +--- +name: forget +description: Delete the Mem0 memories stored for this repository and this user. Use when the user asks to forget, clear, wipe, or delete memories. +disable-model-invocation: true +--- + +# Forget this repository's memories + +This permanently deletes remote memories. Before running anything, tell the +user exactly what will be deleted: their own memories for this repository +only. The repository's project memory is shared by everyone who works in it, +so it stays unless the user explicitly asks to delete that too. + +After the user confirms, run: + +```bash +python3 "${KIMI_PLUGIN_ROOT}/core/memory_cli.py" --harness "kimi" forget --remote --yes +``` + +If the user also asked to delete the repository's shared project memory, add +`--include-project-memory` and say that this removes it for every teammate. + +Report what the command output says was deleted. If the user only wants local +data cleared (evidence log, pending queue), run the same command without +`--remote`. Never pass `--yes` before the user has confirmed in this +conversation. diff --git a/integrations/kimi-plugin/skills/pause/SKILL.md b/integrations/kimi-plugin/skills/pause/SKILL.md new file mode 100644 index 000000000..9f217a468 --- /dev/null +++ b/integrations/kimi-plugin/skills/pause/SKILL.md @@ -0,0 +1,20 @@ +--- +name: pause +description: Pause Mem0 memory capture on this machine. Use when the user wants to stop memories being recorded, for example for private work or experiments. +disable-model-invocation: true +--- + +# Pause memory capture + +To pause (hooks stop capturing and sending session content; a minimal +anonymous telemetry ping still fires at session start unless +`MEM0_TELEMETRY=false`): + +```bash +python3 "${KIMI_PLUGIN_ROOT}/core/memory_cli.py" --harness "kimi" pause +``` + +Confirm the new state back to the user, and remind them that already-created +memories still exist and remain searchable. Pending unsent packets are held +while paused, not expired, and are delivered after resuming. To turn capture +back on, use `/mem0:resume`. diff --git a/integrations/kimi-plugin/skills/remember/SKILL.md b/integrations/kimi-plugin/skills/remember/SKILL.md new file mode 100644 index 000000000..7be5fc9be --- /dev/null +++ b/integrations/kimi-plugin/skills/remember/SKILL.md @@ -0,0 +1,21 @@ +--- +name: remember +description: Acknowledge a "remember this" request and make sure it is captured well. Use when the user explicitly asks to remember, note, or save something for future sessions. +disable-model-invocation: true +--- + +# Remember something for future sessions + +Mem0 creates memories from the session automatically — there is no separate +write command. When the user asks to remember something: + +1. Restate the fact clearly and completely in your reply, in one or two + sentences, including any names, values, or paths it depends on. Your visible + reply is what memory extraction reads, so a precise restatement is what gets + remembered. +2. Tell the user it will be saved with this session's memories when the session + ends or compacts, and that it will surface in future sessions in this + repository (they can check later with /mem0:search). + +Do not invent a storage confirmation or a memory ID — creation happens in the +background after the session. diff --git a/integrations/kimi-plugin/skills/resume/SKILL.md b/integrations/kimi-plugin/skills/resume/SKILL.md new file mode 100644 index 000000000..7dd91be96 --- /dev/null +++ b/integrations/kimi-plugin/skills/resume/SKILL.md @@ -0,0 +1,19 @@ +--- +name: resume +description: Resume Mem0 memory capture after it was paused with /mem0:pause. +disable-model-invocation: true +--- + +# Resume memory capture + +Resume memory capture for this machine. + +Run: + +```bash +python3 "${KIMI_PLUGIN_ROOT}/core/memory_cli.py" --harness "kimi" resume +``` + +Confirm to the user that capture is active again. New sessions record evidence and +create memories as normal; nothing that happened while paused is retroactively +captured. diff --git a/integrations/kimi-plugin/skills/search/SKILL.md b/integrations/kimi-plugin/skills/search/SKILL.md new file mode 100644 index 000000000..eb92a6ca8 --- /dev/null +++ b/integrations/kimi-plugin/skills/search/SKILL.md @@ -0,0 +1,28 @@ +--- +name: search +description: Search memories from earlier Kimi sessions in this repository. Use it when earlier work may already explain the code, error, decision, or command you need, so you can avoid repeating file reads, searches, or experiments. +argument-hint: "[question] [--top-k number] [--category category-name] [--scope repo|dir|mine] [--run-id session-id]" +disable-model-invocation: true +--- + +# Search memories + +Call `search_memories` with the user's question. Treat `--top-k`, `--category`, +`--scope`, and `--run-id` as tool arguments instead of including them in the +query. + +Omit `top_k` to use Mem0's configured default. Omit `category` to search every +category; a category is a best-effort label Mem0 assigned when it saved the +memory, so if a category search misses, repeat it without the category. Omit +`scope` to use the configured default, normally `repo`: this repository's +shared memory, which everyone who works in it contributes to, plus your own +preferences. + +Pass `scope` when the question needs something else: `dir` to narrow the +shared memory to the directory you are working in (a package inside a +monorepo), `mine` for your own preferences alone. + +Pass `run_id` with any scope to retrieve memories saved in a specific coding-agent +session. Omit `run_id` to search across sessions. It filters the memories returned; +it does not identify the session making the search request. Use a known session ID, +never invent one. Return the tool's result directly. diff --git a/integrations/kimi-plugin/skills/status/SKILL.md b/integrations/kimi-plugin/skills/status/SKILL.md new file mode 100644 index 000000000..165c0c675 --- /dev/null +++ b/integrations/kimi-plugin/skills/status/SKILL.md @@ -0,0 +1,23 @@ +--- +name: status +description: Show whether Mem0 memory is working in this repository, covering configuration, capture state, pending flushes, and whether the Mem0 API key is valid. Use when the user asks whether memory is on, why a memory is missing, or anything looks broken. +disable-model-invocation: false +--- + +# Memory status + +Run both commands and report the combined result in plain language: + +```bash +python3 "${KIMI_PLUGIN_ROOT}/core/memory_cli.py" --harness "kimi" status --json +python3 "${KIMI_PLUGIN_ROOT}/core/memory_cli.py" --harness "kimi" doctor +``` + +Summarize, using only fields the JSON actually reports: whether capture is +active or paused, the user ID and repository scope (`repo_id`), whether an +API key is configured, the event/flush/retrieval counts (`flushes` is the +number of completed flushes, not a pending count), and the doctor check +results. If doctor reports an authentication failure (401 / invalid key), say +clearly that the Mem0 API key is invalid or expired and that memories are NOT +being created. Never report an auth failure as "no memories found". Suggest +reinstalling with `--config api_key=...` in that case. diff --git a/integrations/kimi-plugin/tests/test_kimi.py b/integrations/kimi-plugin/tests/test_kimi.py new file mode 100644 index 000000000..ec93be86c --- /dev/null +++ b/integrations/kimi-plugin/tests/test_kimi.py @@ -0,0 +1,190 @@ +from __future__ import annotations + +import importlib.util +import json +import sys +from pathlib import Path + +HOST = Path(__file__).resolve().parents[1] +CORE_ROOT = HOST.parent / "agent-plugin-core" +sys.path.insert(0, str(CORE_ROOT)) + +from build.build import build # noqa: E402 + +SPEC = importlib.util.spec_from_file_location("kimi_adapter", HOST / "hooks" / "adapter.py") +assert SPEC and SPEC.loader +adapter = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(adapter) + + +def _write_kimi_session(home: Path, session_id: str, records: list[dict]) -> Path: + session = home / "sessions" / "wd_mem0_deadbeef1234" / session_id + transcript = session / "agents" / "main" / "wire.jsonl" + transcript.parent.mkdir(parents=True) + transcript.write_text("\n".join(json.dumps(record) for record in records) + "\n", encoding="utf-8") + (home / "session_index.jsonl").write_text( + json.dumps({"sessionId": session_id, "sessionDir": str(session), "workDir": "/work/mem0"}) + "\n", + encoding="utf-8", + ) + return transcript + + +def test_stop_reads_main_response_from_kimi_session_transcript(tmp_path: Path, monkeypatch) -> None: + transcript = _write_kimi_session( + tmp_path, + "session_abc", + [ + {"type": "metadata", "protocol_version": "1.5", "created_at": 1788431400000}, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": {"type": "step.begin", "uuid": "step-1", "turnId": "0", "step": 1}, + "time": 1788431400100, + }, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": { + "type": "content.part", + "stepUuid": "step-1", + "part": {"type": "text", "text": "Earlier answer."}, + }, + "time": 1788431400200, + }, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": {"type": "step.end", "uuid": "step-1", "finishReason": "end_turn"}, + "time": 1788431400300, + }, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": {"type": "step.begin", "uuid": "step-2", "turnId": "1", "step": 1}, + "time": 1788431400400, + }, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": { + "type": "content.part", + "stepUuid": "step-2", + "part": {"type": "think", "think": "Do not capture this reasoning."}, + }, + "time": 1788431400500, + }, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": { + "type": "content.part", + "stepUuid": "step-2", + "part": {"type": "text", "text": "Fixed and "}, + }, + "time": 1788431400600, + }, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": { + "type": "content.part", + "stepUuid": "step-2", + "part": {"type": "text", "text": "tested."}, + }, + "time": 1788431400700, + }, + { + "type": "context.append_loop_event", + "agentId": "main", + "event": {"type": "step.end", "uuid": "step-2", "finishReason": "end_turn"}, + "time": 1788431400800, + }, + ], + ) + monkeypatch.setenv("KIMI_CODE_HOME", str(tmp_path)) + + value = adapter.normalize( + { + "hook_event_name": "Stop", + "session_id": "session_abc", + "session_title": "Fix Kimi capture", + "client_type": "kimi_code_cli", + "cwd": "/work/mem0", + "stop_hook_active": False, + } + ) + + assert value["last_assistant_message"] == "Fixed and tested." + assert value["transcript_path"] == str(transcript) + + +def test_repeated_sidekick_invocations_get_distinct_run_ids(tmp_path: Path) -> None: + store = adapter.hook_runner.EvidenceStore(tmp_path / "evidence.sqlite3") + payload = { + "hook_event_name": "SubagentStart", + "session_id": "session_abc", + "session_title": "Fix Kimi capture", + "client_type": "kimi_code_cli", + "cwd": str(tmp_path), + "agent_name": "sidekick", + "prompt": "Fix the adapter", + } + + try: + adapter._sidekick_start(store, adapter.normalize({**payload, "agent_id": "first"})) + adapter._sidekick_start(store, adapter.normalize({**payload, "agent_id": "second"})) + adapter._sidekick_stop( + store, + adapter.normalize( + { + **payload, + "hook_event_name": "SubagentStop", + "response": "First run complete.", + "agent_id": "first", + } + ), + ) + adapter._sidekick_stop( + store, + adapter.normalize( + { + **payload, + "hook_event_name": "SubagentStop", + "response": "Second run complete.", + "agent_id": "second", + } + ), + ) + + rows = store.conn.execute( + "SELECT agent_id, stopped_at, final_message FROM sidekick_runs ORDER BY started_at, agent_id" + ).fetchall() + finally: + store.close() + + assert len(rows) == 2 + assert rows[0]["agent_id"] != rows[1]["agent_id"] + assert all(row["stopped_at"] for row in rows) + assert {row["agent_id"]: row["final_message"] for row in rows} == { + "first": "First run complete.", "second": "Second run complete." + } + + +def test_native_kimi_bundle_uses_inline_native_contract(tmp_path: Path) -> None: + root = build("kimi", "native", tmp_path / "kimi") + + manifest = json.loads((root / "kimi.plugin.json").read_text(encoding="utf-8")) + assert manifest["skills"] == "./skills/" + assert manifest["agents"] == "./agents/" + assert manifest["mcpServers"]["mem0"]["args"] == ["./core/mcp_server.py"] + assert {hook["event"] for hook in manifest["hooks"]} >= { + "SessionStart", + "UserPromptSubmit", + "PostToolUse", + "SubagentStart", + "SubagentStop", + "PreCompact", + "SessionEnd", + } + assert (root / "agents" / "sidekick.md").is_file() + assert not any(path.is_symlink() for path in root.rglob("*")) diff --git a/integrations/mem0-agent-plugin/core/mcp_server.py b/integrations/mem0-agent-plugin/core/mcp_server.py new file mode 100644 index 000000000..036fbbdc9 --- /dev/null +++ b/integrations/mem0-agent-plugin/core/mcp_server.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Expose Mem0's memory search as one local coding-agent tool.""" + +from __future__ import annotations + +import json +import os +import sys +from typing import Any + +import telemetry +from memory_core import ( + CODING_MEMORY_CATEGORY_NAMES, + PLUGIN_VERSION, + SEARCH_SCOPES, + format_search_result, + resolve_repo, + search_memories, +) + +PROTOCOL_VERSION = "2024-11-05" +TOOL_NAME = "search_memories" +TOOL_DESCRIPTION = ( + "Search memories from earlier work in this repository. ALWAYS call this " + "tool before answering anything that could depend on prior context: the " + "user's preferences, facts about this codebase, history, people, projects, " + "or earlier decisions. Do not rely on the chat window alone. The " + "repository's memory is shared by everyone who works in it and includes " + "what it took to run, test, or build here, so search before assuming an " + "invocation works. The scope argument changes what is searched: 'repo' " + "(default) is the whole repository's shared memory plus your own " + "preferences, 'dir' narrows the shared part to the directory you are " + "working in, and 'mine' is your preferences alone." +) +TOOL_SCHEMA = { + "type": "object", + "properties": { + "query": { + "type": "string", + "minLength": 1, + "maxLength": 2000, + "description": "A direct question about earlier work in this repository.", + }, + "top_k": { + "type": "integer", + "minimum": 1, + "maximum": 20, + "description": "Maximum memories to return. Uses Mem0's configured default when omitted.", + }, + "category": { + "type": "string", + "enum": list(CODING_MEMORY_CATEGORY_NAMES), + "description": "Optional memory category. Omit to search every category.", + }, + "scope": { + "type": "string", + "enum": list(SEARCH_SCOPES), + "description": ( + "Which memories to search. 'repo' (default) is the whole repository's " + "shared memory plus your own preferences, 'dir' narrows the shared " + "part to the current directory, 'mine' is your preferences alone." + ), + }, + "run_id": { + "type": "string", + "minLength": 1, + "description": ( + "Optional coding-agent session ID. With any scope, restricts results to memories " + "saved in that session. Omit to recall memories across sessions." + ), + }, + }, + "required": ["query"], + "additionalProperties": False, +} + + +class ToolInputError(ValueError): + pass + + +def _validate_arguments( + arguments: Any, +) -> tuple[str, int | None, str | None, str | None, str | None]: + if not isinstance(arguments, dict): + raise ToolInputError("Search arguments must be an object.") + + unknown = set(arguments) - {"query", "top_k", "category", "scope", "run_id"} + if unknown: + raise ToolInputError(f"Unknown search argument: {sorted(unknown)[0]}") + + query = arguments.get("query") + if not isinstance(query, str) or not query.strip(): + raise ToolInputError("query must be a non-empty string.") + query = query.strip() + if len(query) > 2000: + raise ToolInputError("query must be at most 2,000 characters.") + + top_k = arguments.get("top_k") + if top_k is not None and ( + isinstance(top_k, bool) or not isinstance(top_k, int) or not 1 <= top_k <= 20 + ): + raise ToolInputError("top_k must be an integer from 1 to 20.") + + category = arguments.get("category") + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ToolInputError("category must be one of Mem0's supported categories.") + + scope = arguments.get("scope") + if scope is not None and scope not in SEARCH_SCOPES: + raise ToolInputError(f"scope must be one of {list(SEARCH_SCOPES)}.") + + run_id = arguments.get("run_id") + if run_id is not None: + if not isinstance(run_id, str) or not run_id.strip(): + raise ToolInputError("run_id must be a non-empty string.") + run_id = run_id.strip() + + return query, top_k, category, scope, run_id + + +def call_search_memories(arguments: Any, cwd: str | None = None) -> str: + query, top_k, category, scope, run_id = _validate_arguments(arguments) + repo = resolve_repo(cwd or os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + result = search_memories( + None, + repo, + None, + query, + top_k=top_k, + category=category, + scope=scope, + run_id=run_id, + operation="mcp-search", + ) + return format_search_result(result) + + +def _workspace_cwd(params: dict[str, Any]) -> str | None: + meta = params.get("_meta") + if not isinstance(meta, dict): + return None + metadata = meta.get("x-codex-turn-metadata") + if not isinstance(metadata, dict): + return None + workspaces = metadata.get("workspaces") or {} + if isinstance(workspaces, dict): + return next((path for path in workspaces if isinstance(path, str) and path), None) + return None + + +def _tool_response(text: str, *, is_error: bool = False) -> dict[str, Any]: + return { + "content": [{"type": "text", "text": text}], + "isError": is_error, + } + + +def handle_request(message: Any) -> dict[str, Any] | None: + if not isinstance(message, dict): + return None + request_id = message.get("id") + method = message.get("method") + + if method == "notifications/initialized": + return None + if method == "initialize": + requested = (message.get("params") or {}).get("protocolVersion") + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "protocolVersion": requested or PROTOCOL_VERSION, + "capabilities": {"tools": {"listChanged": False}}, + "serverInfo": {"name": "mem0", "version": PLUGIN_VERSION}, + }, + } + if method == "ping": + return {"jsonrpc": "2.0", "id": request_id, "result": {}} + if method == "tools/list": + return { + "jsonrpc": "2.0", + "id": request_id, + "result": { + "tools": [ + { + "name": TOOL_NAME, + "description": TOOL_DESCRIPTION, + "inputSchema": TOOL_SCHEMA, + "annotations": { + "readOnlyHint": True, + "idempotentHint": True, + "openWorldHint": True, + }, + } + ] + }, + } + if method == "tools/call": + params = message.get("params") or {} + if params.get("name") != TOOL_NAME: + result = _tool_response("Unknown Mem0 tool.", is_error=True) + else: + try: + result = _tool_response( + call_search_memories(params.get("arguments"), _workspace_cwd(params)) + ) + except ToolInputError as exc: + result = _tool_response(str(exc), is_error=True) + except Exception: + result = _tool_response("Memory search failed.", is_error=True) + return {"jsonrpc": "2.0", "id": request_id, "result": result} + if request_id is None: + return None + return { + "jsonrpc": "2.0", + "id": request_id, + "error": {"code": -32601, "message": "Method not found"}, + } + + +def main() -> int: + for raw_line in sys.stdin: + try: + message = json.loads(raw_line) + response = handle_request(message) + except json.JSONDecodeError: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32700, "message": "Parse error"}, + } + except Exception: + response = { + "jsonrpc": "2.0", + "id": None, + "error": {"code": -32603, "message": "Internal error"}, + } + if response is not None: + sys.stdout.write(json.dumps(response, separators=(",", ":")) + "\n") + sys.stdout.flush() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/mem0-agent-plugin/core/memory_cli.py b/integrations/mem0-agent-plugin/core/memory_cli.py new file mode 100644 index 000000000..595729feb --- /dev/null +++ b/integrations/mem0-agent-plugin/core/memory_cli.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Mem0 diagnostics and user controls.""" + +from __future__ import annotations + +import argparse +import json +import os + +import telemetry +from memory_core import ( + EvidenceStore, + api_key, + data_dir, + doctor, + forget_remote_repo, + configure_harness, + resolve_repo, + user_id, +) + + +def _print_status(value: dict) -> None: + last = value.get("last_operation") or {} + print(f"Mem0: {'paused' if value['paused'] else 'active'}") + print(f"Repository: {value['repo_id']}") + print(f"Local data: {value['data_dir']}") + print(f"API key: {'configured' if value['api_key_configured'] else 'missing'}") + print( + "Saved on this computer: " + f"{value['events']} session details, {value['flushes']} memory updates" + ) + print( + f"Used in this repository: {value['retrievals']} memories returned, " + f"{value['sidekick_runs']} sidekick runs" + ) + if last: + item_label = "" + if last["operation"] in {"flush", "flush-retry"}: + item_label = f", {last['item_count']} memories" + operation = ( + "memory update" + if last["operation"] in {"flush", "flush-retry"} + else last["operation"].replace("-", " ") + ) + print( + f"Last {operation}: " + f"{'succeeded' if last['success'] else 'failed'} " + f"({last['duration_ms']:.1f} ms{item_label})" + ) + sidekick = value.get("last_sidekick") or {} + if sidekick: + state = "finished" if sidekick.get("stopped_at") else "started" + print( + "Last sidekick: " + f"{state}, received {sidekick['context_chars']} characters of memory, " + f"agent {sidekick['agent_id']}" + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--plugin-data-dir", default="") + parser.add_argument("--harness", default="") + subparsers = parser.add_subparsers(dest="command", required=True) + + status = subparsers.add_parser("status") + status.add_argument("--json", action="store_true") + + doctor_parser = subparsers.add_parser("doctor") + doctor_parser.add_argument("--json", action="store_true") + + subparsers.add_parser("pause") + subparsers.add_parser("resume") + + forget = subparsers.add_parser("forget") + forget.add_argument("--remote", action="store_true") + forget.add_argument("--yes", action="store_true") + forget.add_argument("--include-project-memory", action="store_true") + + args = parser.parse_args() + if args.harness: + source_tag = f"{args.harness.replace('-', '_')}_plugin" + configure_harness(args.harness, source_tag=source_tag) + telemetry.init(harness=args.harness, source_tag=source_tag.upper()) + if args.plugin_data_dir: + os.environ["MEM0_CODE_DATA_DIR"] = args.plugin_data_dir + store = EvidenceStore() + try: + repo = resolve_repo(os.getcwd()) + telemetry.record("control", repo=repo, action=args.command) + if args.command == "status": + result = { + **store.status(repo.identity), + "repo_id": repo.identity, + "app_id": repo.app_id, + "project_id": repo.project_id, + "directory": repo.directory, + "user_id": user_id(), + "data_dir": str(data_dir()), + "api_key_configured": bool(api_key()), + } + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + _print_status(result) + elif args.command == "doctor": + result = doctor(os.getcwd()) + if args.json: + print(json.dumps(result, indent=2, default=str)) + else: + for name, check in result["checks"].items(): + print( + f"{'PASS' if check['ok'] else 'FAIL'} {name}: {check['detail']}" + ) + return 0 if result["ok"] else 1 + elif args.command == "pause": + store.set_setting("paused", "true") + print("Mem0 stopped saving and searching memories.") + elif args.command == "resume": + store.set_setting("paused", "false") + print("Mem0 resumed saving and searching memories.") + elif args.command == "forget": + if not args.yes: + print( + "Refusing to delete data without --yes. Add --remote to also " + "delete this user/repository scope from Mem0." + ) + return 2 + remote_result = ( + forget_remote_repo( + repo, include_project_memory=args.include_project_memory + ) + if args.remote + else None + ) + local_result = store.forget_local_repo(repo.identity) + print( + json.dumps( + {"local": local_result, "remote": remote_result}, + indent=2, + default=str, + ) + ) + if remote_result and remote_result.get("status") == "error": + return 1 + finally: + store.close() + telemetry.spawn_flush() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/integrations/mem0-agent-plugin/core/memory_core.py b/integrations/mem0-agent-plugin/core/memory_core.py new file mode 100644 index 000000000..cf71196b8 --- /dev/null +++ b/integrations/mem0-agent-plugin/core/memory_core.py @@ -0,0 +1,2645 @@ +#!/usr/bin/env python3 +"""Shared core for Mem0 agent plugins. + +Hooks record small session details locally. When the agent compacts or ends the +session, Mem0 sends the useful parts to the platform so it can create memories. +The agent can search those memories during later work in the repository. +""" + +from __future__ import annotations + +import functools +import hashlib +import json +import math +import os +import re +import sqlite3 +import subprocess +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Iterable + +import telemetry + +DEFAULT_API_URL = "https://api.mem0.ai" +PLUGIN_VERSION = "0.3.1" + +_harness_name: str = "generic" +_harness_env_prefix: str = "MEM0_PLUGIN" +_harness_data_dir_name: str = "mem0-plugin" +_harness_source_tag: str = "mem0_plugin" + + +def configure_harness( + name: str, + env_prefix: str = "", + data_dir_name: str = "", + source_tag: str = "", +) -> None: + global _harness_name, _harness_env_prefix, _harness_data_dir_name, _harness_source_tag + _harness_name = name + _harness_env_prefix = env_prefix or f"MEM0_{name.upper().replace('-', '_')}" + _harness_data_dir_name = data_dir_name or f"{name}-plugin" + _harness_source_tag = source_tag or f"{name.replace('-', '_')}_plugin" + + +def harness_config() -> dict[str, str]: + return { + "name": _harness_name, + "env_prefix": _harness_env_prefix, + "data_dir_name": _harness_data_dir_name, + "source_tag": _harness_source_tag, + } + + +MAX_COMMAND_CHARS = 2000 +MAX_RESULT_CHARS = 2500 +MAX_EPISODE_CHARS = 12000 +CHECKPOINT_EXCHANGES = 5 +CHECKPOINT_MESSAGES = 10 +CHECKPOINT_SOURCE_CHARS = 40000 +DEFAULT_MAX_CONTEXT_CHARS = 4000 +MAX_EXTRACTION_INPUT_TOKENS = 24000 +MAX_FLUSH_ATTEMPTS = 5 +FORGET_PAGE_SIZE = 100 +FORGET_MAX_PAGES = 50 + +PROJECT_MEMORY_INSTRUCTIONS = """Save concise repository facts that will help anyone with future coding work in this repository. + +A completed change should produce one memory explaining the resulting behavior, where it is implemented when useful, and any important constraints or reasoning. Exploration or accepted decisions may produce separate memories only when they are independently useful. + +A command that failed and was then made to work should produce one memory naming the failing invocation, the error it returned, and the invocation that succeeded. Do not save one-off errors caused by an edit still in progress, transient network failures, or anything a rerun would fix on its own. + +Use the current coding agent's final response for conclusions about current repository behavior. Do not save proposed or recommended changes unless the user accepted them or the coding agent completed them. Treat subagent responses as supporting repository evidence, not as decisions. + +Write about the repository, not the user, assistant, session, or task. Do not save personal preferences. Do not save a memory that only states which repository, branch, or directory the session worked in. Do not include test results, documentation updates, release notes, or temporary state. + +If nothing useful was established, return no memories.""" + +PERSONAL_MEMORY_INSTRUCTIONS = """Save concise facts about the user that will help in any repository: preferred tools, package managers, languages, coding style, review and communication preferences, and anything the user explicitly asked to be remembered about themselves. + +Write in the third person about the user, not about the repository, the assistant, the session, or the task. Do not save repository facts, project decisions, commands, or what was built. + +Never save that the user has no preferences or that nothing was learned. If nothing was learned about the user, return no memories.""" + +CODING_MEMORY_CATEGORIES = [ + { + "project_knowledge": ( + "What the project is and how its code, APIs, data, files, and " + "components work." + ) + }, + { + "decisions_and_constraints": ( + "Why an approach was chosen, what must remain true, and rules future " + "work must follow." + ) + }, + { + "workflows": ( + "How to run, test, debug, deploy, configure, or otherwise work on the " + "project." + ) + }, + { + "problems_and_fixes": ( + "Bugs, failures, known pitfalls, their causes, and how to fix or avoid " + "them." + ) + }, + { + "results": ( + "Outcomes and measurements from tests, benchmarks, experiments, or " + "investigations." + ) + }, +] +CODING_MEMORY_CATEGORY_NAMES = tuple( + category_name + for category in CODING_MEMORY_CATEGORIES + for category_name in category +) + +TEST_COMMAND_RE = re.compile( + r"(?:^|\s)(?:pytest|py\.test|jest|vitest|go\s+test|cargo\s+test|" + r"npm\s+(?:run\s+)?test|pnpm\s+(?:run\s+)?test|yarn\s+test|" + r"mvn\s+test|gradle\s+test|make\s+test)(?:\s|$)", + re.IGNORECASE, +) +BUILD_COMMAND_RE = re.compile( + r"(?:^|\s)(?:npm|pnpm|yarn)\s+(?:run\s+)?build(?:\s|$)|" + r"(?:^|\s)(?:cargo|go|mvn|gradle|make)\s+build(?:\s|$)", + re.IGNORECASE, +) + +SECRET_PATTERNS = [ + re.compile(r"(?i)(authorization\s*[:=]\s*(?:bearer|token)\s+)[^\s\"']+"), + re.compile( + r"(?i)((?:api[_-]?key|secret[_-]?access[_-]?key|session[_-]?token)\s*[:=]\s*)[^\s\"']+" + ), + re.compile( + r"(?i)((?:access[_-]?token|refresh[_-]?token|password|credential)" + r"\s*[:=]\s*)[^\s&\"']+" + ), + re.compile(r"\b(?:sk|m0|mem0_sk|psk)-[A-Za-z0-9_\-]{12,}\b"), + re.compile(r"\b(?:ASIA|AKIA)[A-Z0-9]{12,}\b"), + re.compile(r"\b(?:ghp_|github_pat_|xox[baprs]-)[A-Za-z0-9_\-]{12,}\b"), + re.compile( + r"-----BEGIN [^-]*PRIVATE KEY-----.*?-----END [^-]*PRIVATE KEY-----", + re.DOTALL, + ), + re.compile( + r'(?i)("(?:api[_-]?key|password|secret(?:[_-]?access[_-]?key)?' + r'|(?:access|refresh|session)[_-]?token|token|authorization|credential' + r')"\s*:\s*")(?:\\.|[^"\\])*' + ), +] + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def redact(value: Any) -> str: + text = ( + value + if isinstance(value, str) + else json.dumps(value, ensure_ascii=False, default=str) + ) + for pattern in SECRET_PATTERNS: + if pattern.groups: + text = pattern.sub(r"\1[REDACTED]", text) + else: + text = pattern.sub("[REDACTED]", text) + return text + + +def bounded(value: Any, limit: int) -> str: + text = redact(value).strip() + if len(text) <= limit: + return text + return text[:limit] + f"\n...[truncated {len(text) - limit} chars]" + + +def _git(cwd: str, *args: str) -> str: + try: + result = subprocess.run( + ["git", "-C", cwd, *args], + check=False, + capture_output=True, + text=True, + timeout=0.5, + ) + except (OSError, subprocess.TimeoutExpired): + return "" + return result.stdout.strip() if result.returncode == 0 else "" + + +def _normalize_remote(remote: str) -> str: + remote = remote.strip() + if remote.startswith("git@") and ":" in remote: + host_path = remote[4:].replace(":", "/", 1) + remote = f"https://{host_path}" + if remote.endswith(".git"): + remote = remote[:-4] + if "://" in remote: + parsed = urllib.parse.urlsplit(remote) + hostname = parsed.hostname or "" + if parsed.port: + hostname = f"{hostname}:{parsed.port}" + remote = urllib.parse.urlunsplit( + (parsed.scheme, hostname, parsed.path, parsed.query, parsed.fragment) + ) + return remote.rstrip("/") + + +_WILDCARD_SCOPE = re.compile(r"^\*+$") + + +def _scope_value(raw: str | None) -> str: + """Reject wildcards as identities: they are filter syntax and would widen the scope.""" + value = (raw or "").strip() + return "" if _WILDCARD_SCOPE.match(value) else value + + +SEARCH_SCOPES = ("repo", "dir", "mine") +DEFAULT_SEARCH_SCOPE = "repo" + +def directory_app_id(repo: RepoContext) -> str: + """The app_id of the directory this session runs in: the repository at the root, repository/path below it.""" + return f"{repo.app_id}/{repo.directory}" if repo.directory else repo.app_id + + +def directory_chain(repo: RepoContext) -> list[str]: + """Every directory a memory belongs to, from the top-level folder down to the one it was written in.""" + parts = repo.directory.split("/") if repo.directory else [] + return ["/".join(parts[: index + 1]) for index in range(len(parts))] + + +def _shared_project_ids(repo: RepoContext) -> list[str]: + """Current and pre-upgrade namespaces, shared by recall and explicit deletion.""" + if not repo.identity.startswith("local:") and repo.project_id != repo.app_id: + return [repo.project_id, repo.app_id] + return [repo.project_id] + + +def _search_filters(user: str, repo: RepoContext, scope: str) -> dict[str, Any]: + """Build the scope filter: app_id scopes to the repo, then union shared and personal lanes.""" + app_scope = {"app_id": repo.app_id} + mine = {"AND": [{"user_id": user}, app_scope]} + if scope == "mine": + return mine + projects = [{"AND": [{"agent_id": project_id}, app_scope]} for project_id in _shared_project_ids(repo)] + shared: dict[str, Any] = projects[0] if len(projects) == 1 else {"OR": projects} + if scope == "dir" and repo.directory: + shared = {"AND": [shared, {"metadata": {"dirs": {"contains": repo.directory}}}]} + return {"OR": [shared, mine]} + + +def search_scope() -> str: + configured = ( + _plugin_option("search_scope", "MEM0_CODE_SEARCH_SCOPE") or "" + ).strip().lower() + return configured if configured in SEARCH_SCOPES else DEFAULT_SEARCH_SCOPE + + +def resolve_search_scope(scope: str | None) -> str: + value = (scope or search_scope()).strip().lower() + if value not in SEARCH_SCOPES: + raise ValueError(f"Unknown search scope: {value}") + return value + + +def _legacy_project_map(cwd: str, root: str, raw_remote: str) -> str: + """Return the project name used by the previous Claude Code plugin.""" + try: + data = json.loads((Path.home() / ".mem0" / "project_map.json").read_text()) + except (OSError, json.JSONDecodeError): + return "" + if not isinstance(data, dict): + return "" + + keys = list(dict.fromkeys([cwd, root, os.path.realpath(cwd), os.path.realpath(root)])) + if raw_remote: + keys.append(f"remote:{hashlib.sha256(raw_remote.encode()).hexdigest()[:16]}") + for key in keys: + value = data.get(key) + if isinstance(value, str) and value.strip(): + return value.strip() + return "" + + +def _legacy_project_id(cwd: str, root: str, raw_remote: str, identity: str) -> str: + """Use the repository namespace created by the previous Mem0 plugin.""" + configured = _scope_value(os.environ.get("MEM0_PROJECT_ID")) + if configured: + return configured + + mapped = _scope_value(_legacy_project_map(cwd, root, raw_remote)) + if mapped: + return mapped + + remote = raw_remote or ("" if identity.startswith("local:") else identity) + remote = remote.strip().removesuffix(".git") + for prefix in ("https://", "http://", "ssh://", "git://"): + if remote.startswith(prefix): + remote = remote[len(prefix) :] + break + else: + remote = re.sub(r"^git@", "", remote) + parts = [part for part in remote.replace(":", "/", 1).split("/") if part] + if len(parts) >= 2: + return f"{parts[-2]}-{parts[-1]}".replace("/", "-").replace(":", "-") + if parts: + return parts[-1].replace("/", "-").replace(":", "-") + return os.path.basename(root or cwd) or "unknown" + + +@dataclass(frozen=True) +class RepoContext: + cwd: str + root: str + identity: str + app_id: str + branch: str + head_sha: str + project_id: str = "" + directory: str = "" + + +def _project_id(root: str, identity: str, app_id: str) -> str: + """The shared namespace: includes a host hash so repos with the same owner/name on different hosts stay apart.""" + if not identity.startswith("local:"): + return f"{app_id}-{hashlib.sha256(identity.encode()).hexdigest()[:10]}" + return f"local-{app_id}-{hashlib.sha256(root.encode()).hexdigest()[:10]}" + + +def _relative_directory(cwd: str, root: str) -> str: + relative = os.path.relpath(cwd, root) + return "" if relative == "." or relative.startswith("..") else relative.replace(os.sep, "/") + + +@dataclass(frozen=True) +class MemorySearchResult: + succeeded: bool + matched_count: int + already_shown_count: int + memories: list[dict[str, Any]] + + +@functools.lru_cache(maxsize=64) +def _resolve_repo_cached(cwd: str) -> RepoContext: + given_cwd = cwd + cwd = os.path.realpath(cwd) + given_root = _git(cwd, "rev-parse", "--show-toplevel") or given_cwd + root = os.path.realpath(given_root) + raw_remote = _git(root, "config", "--get", "remote.origin.url") + remote = _normalize_remote(raw_remote) + identity = remote or f"local:{root}" + app_id = _legacy_project_id(given_cwd, given_root, raw_remote, identity) + return RepoContext( + cwd=cwd, + root=root, + identity=identity, + app_id=app_id, + branch=_git(root, "branch", "--show-current") or "detached", + head_sha=_git(root, "rev-parse", "HEAD"), + project_id=_project_id(root, identity, app_id), + directory=_relative_directory(cwd, root), + ) + + +def resolve_repo(cwd: str | None) -> RepoContext: + return _resolve_repo_cached(os.path.abspath(cwd or os.getcwd())) + + +def api_key() -> str: + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return configured + try: + return (data_dir() / "api-key").read_text(encoding="utf-8").strip() + except OSError: + return "" + + +def cache_plugin_api_key() -> bool: + """Bridge host's hook-only sensitive config into plugin-owned storage.""" + configured = ( + os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if not configured: + return False + + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + path = directory / "api-key" + temporary = directory / f"api-key.{os.getpid()}.tmp" + descriptor = os.open( + temporary, + os.O_WRONLY | os.O_CREAT | os.O_TRUNC, + 0o600, + ) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + handle.write(configured) + os.replace(temporary, path) + os.chmod(path, 0o600) + finally: + try: + temporary.unlink() + except FileNotFoundError: + pass + return True + + +def clear_stale_api_key_cache() -> bool: + """Drop the cached key file once every configured key source is gone.""" + configured = ( + os.environ.get("MEM0_API_KEY") + or os.environ.get("PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY") + or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") + or "" + ).strip() + if configured: + return False + path = data_dir() / "api-key" + if not path.exists(): + return False + try: + path.unlink() + except OSError: + return False + return True + + +def detached_process_kwargs(platform: str | None = None) -> dict: + """Keep a spawned worker alive after the coding agent exits, on POSIX and Windows.""" + if (platform or sys.platform) == "win32": + return { + "creationflags": subprocess.DETACHED_PROCESS + | subprocess.CREATE_NEW_PROCESS_GROUP + } + return {"start_new_session": True} + + +def _plugin_option(name: str, fallback: str = "") -> str: + return ( + os.environ.get(f"PLUGIN_OPTION_{name.upper()}") + or os.environ.get(f"CLAUDE_PLUGIN_OPTION_{name.upper()}") + or os.environ.get(fallback) + or "" + ).strip() + + +def user_id() -> str: + return ( + _scope_value(_plugin_option("user_id", "MEM0_CODE_USER_ID")) + or _scope_value(os.environ.get("MEM0_USER_ID")) + or _scope_value(os.environ.get("MEM0_RESOLVED_USER_ID")) + or _scope_value(os.environ.get("USER")) + or _scope_value(os.environ.get("USERNAME")) + or "default" + ) + + +def data_dir() -> Path: + configured = ( + os.environ.get("MEM0_CODE_DATA_DIR") + or os.environ.get("MEM0_PLUGIN_DATA_DIR") + or os.environ.get("PLUGIN_DATA") + or os.environ.get("CLAUDE_PLUGIN_DATA") + ) + return ( + Path(configured).expanduser() if configured else Path.home() / ".mem0" / _harness_data_dir_name + ) + + +def _bool_option(name: str, fallback: str, default: bool = False) -> bool: + value = _plugin_option(name, fallback) + if not value: + return default + return value.lower() in {"1", "true", "yes", "on"} + + +def _int_option(name: str, fallback: str, default: int) -> int: + value = _plugin_option(name, fallback) + try: + return int(value) if value else default + except ValueError: + return default + + + +def _checkpoint_message(event: dict[str, Any]) -> str: + kind = event.get("kind") + payload = event.get("payload") or {} + if kind == "user_prompt": + return redact(payload.get("text", "")).strip() + if kind == "assistant_stop": + transcript_messages = payload.get("transcript_messages") or [] + if isinstance(transcript_messages, list): + text = "\n".join( + str(message.get("content") or "") + for message in transcript_messages + if isinstance(message, dict) and message.get("content") + ) + if text: + return text + return redact(payload.get("text", "")).strip() + if kind == "sidekick_stop": + return redact(payload.get("final_message", "")).strip() + return "" + + +def checkpoint_stats(events: list[dict[str, Any]]) -> tuple[int, int, int]: + """Return completed exchanges, messages, and source characters.""" + completed = sum(event.get("kind") == "assistant_stop" for event in events) + contents = [content for event in events if (content := _checkpoint_message(event))] + return completed, len(contents), sum(len(content) for content in contents) + + +def select_checkpoint_events( + events: list[dict[str, Any]], *, force: bool +) -> list[dict[str, Any]]: + """Select one ordered extraction block without splitting an exchange.""" + for index, event in enumerate(events): + if event.get("kind") != "assistant_stop": + continue + candidate = events[: index + 1] + completed, messages, source_chars = checkpoint_stats(candidate) + if ( + completed >= CHECKPOINT_EXCHANGES + or messages >= CHECKPOINT_MESSAGES + or source_chars >= CHECKPOINT_SOURCE_CHARS + ): + return candidate + return events if force else [] + + +class EvidenceStore: + def __init__(self, path: Path | None = None): + directory = data_dir() if path is None else path.parent + directory.mkdir(parents=True, exist_ok=True) + self.path = path or directory / "evidence.sqlite3" + try: + self._open() + except sqlite3.DatabaseError: + self._quarantine() + self._open() + + def _open(self) -> None: + self.conn = sqlite3.connect(self.path, timeout=10) + self.conn.row_factory = sqlite3.Row + try: + self.conn.execute("PRAGMA journal_mode=WAL") + self.conn.execute("PRAGMA busy_timeout=10000") + self._migrate() + except sqlite3.DatabaseError: + self.conn.close() + raise + + def _quarantine(self) -> None: + """Move an unreadable database aside so capture restarts cleanly.""" + stamp = int(time.time()) + for suffix in ("", "-wal", "-shm"): + source = Path(f"{self.path}{suffix}") + try: + source.replace(f"{self.path}.corrupt-{stamp}{suffix}") + except FileNotFoundError: + continue + except OSError: + try: + source.unlink() + except OSError: + pass + telemetry.record("db_quarantined") + + def close(self) -> None: + self.conn.close() + + def _migrate(self) -> None: + self.conn.executescript( + """ + CREATE TABLE IF NOT EXISTS events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + created_at TEXT NOT NULL, + kind TEXT NOT NULL, + payload_json TEXT NOT NULL, + flush_id TEXT + ); + CREATE INDEX IF NOT EXISTS events_session_idx + ON events(repo_id, session_id, flush_id, id); + + CREATE TABLE IF NOT EXISTS session_scopes ( + session_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + root TEXT NOT NULL, + branch TEXT NOT NULL, + head_sha TEXT NOT NULL, + created_at TEXT NOT NULL, + directory TEXT NOT NULL DEFAULT '' + ); + + CREATE TABLE IF NOT EXISTS flushes ( + packet_id TEXT PRIMARY KEY, + repo_id TEXT NOT NULL, + app_id TEXT NOT NULL, + session_id TEXT NOT NULL, + reason TEXT NOT NULL, + event_start INTEGER NOT NULL, + event_end INTEGER NOT NULL, + status TEXT NOT NULL, + episode_event_id TEXT, + semantic_event_id TEXT, + error TEXT, + attempts INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + CREATE TABLE IF NOT EXISTS retrievals ( + session_id TEXT NOT NULL, + repo_id TEXT NOT NULL, + memory_id TEXT NOT NULL, + injected_at TEXT NOT NULL, + rank INTEGER, + score REAL, + memory_text TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(session_id, repo_id, memory_id) + ); + + CREATE TABLE IF NOT EXISTS operations ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + created_at TEXT NOT NULL, + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + operation TEXT NOT NULL, + duration_ms REAL NOT NULL, + success INTEGER NOT NULL, + item_count INTEGER NOT NULL DEFAULT 0, + request_chars INTEGER NOT NULL DEFAULT 0, + response_chars INTEGER NOT NULL DEFAULT 0, + error TEXT + ); + + CREATE TABLE IF NOT EXISTS sidekick_runs ( + repo_id TEXT NOT NULL, + session_id TEXT NOT NULL, + agent_id TEXT NOT NULL, + agent_type TEXT NOT NULL, + started_at TEXT NOT NULL, + stopped_at TEXT, + transcript_path TEXT, + context_chars INTEGER NOT NULL DEFAULT 0, + final_message TEXT, + PRIMARY KEY(repo_id, session_id, agent_id) + ); + CREATE INDEX IF NOT EXISTS sidekick_runs_repo_idx + ON sidekick_runs(repo_id, started_at); + + CREATE TABLE IF NOT EXISTS settings ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL, + updated_at TEXT NOT NULL + ); + + """ + ) + # Remove the pre-0.1.1 no-tools snapshot implementation. The real coding + # sidekick is a native Claude Code agent and stores no state in this DB. + self.conn.executescript( + """ + DROP TABLE IF EXISTS sidekick_calls; + DROP TABLE IF EXISTS sidekick_snapshots; + DROP TABLE IF EXISTS sidekick_state; + DROP TABLE IF EXISTS sidekick_packets; + """ + ) + self._ensure_column("retrievals", "rank", "INTEGER") + self._ensure_column("retrievals", "score", "REAL") + self._ensure_column("retrievals", "memory_text", "TEXT") + self._ensure_column("retrievals", "context_chars", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("flushes", "attempts", "INTEGER NOT NULL DEFAULT 0") + self._ensure_column("session_scopes", "directory", "TEXT NOT NULL DEFAULT ''") + self.conn.commit() + + def _ensure_column(self, table: str, column: str, declaration: str) -> None: + columns = { + str(row["name"]) + for row in self.conn.execute(f"PRAGMA table_info({table})").fetchall() + } + if column not in columns: + self.conn.execute(f"ALTER TABLE {table} ADD COLUMN {column} {declaration}") + + def record_event( + self, + repo: RepoContext, + session_id: str, + kind: str, + payload: dict[str, Any], + ) -> int: + cursor = self.conn.execute( + """INSERT INTO events + (repo_id, app_id, session_id, created_at, kind, payload_json) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + repo.app_id, + session_id, + utc_now(), + kind, + json.dumps(payload, ensure_ascii=False, sort_keys=True), + ), + ) + self.conn.commit() + return int(cursor.lastrowid) + + def record_assistant_response(self, repo: RepoContext, session_id: str, message: str) -> None: + """Ignore repeated response hooks until another prompt or a different answer arrives.""" + with self.conn: + # Serialize the check and insert across concurrent Stop and SessionEnd hooks. + self.conn.execute("BEGIN IMMEDIATE") + previous = self.conn.execute( + """SELECT kind, payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind IN ('user_prompt', 'assistant_stop') + ORDER BY id DESC LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if ( + previous is not None + and previous["kind"] == "assistant_stop" + and json.loads(previous["payload_json"]).get("text") == message + ): + return + self.record_event(repo, session_id, "assistant_stop", {"text": message}) + + def repo_for_session(self, session_id: str, cwd: str | None) -> RepoContext: + """Keep one project scope for every hook in a coding-agent session.""" + current = resolve_repo(cwd) + if session_id == "unknown-session": + return current + + with self.conn: + self.conn.execute( + """INSERT OR IGNORE INTO session_scopes + (session_id, repo_id, app_id, root, branch, head_sha, created_at, directory) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + current.identity, + current.app_id, + current.root, + current.branch, + current.head_sha, + utc_now(), + current.directory, + ), + ) + scope = self.conn.execute( + "SELECT * FROM session_scopes WHERE session_id = ?", (session_id,) + ).fetchone() + same_git_repo = ( + current.identity == scope["repo_id"] and bool(current.head_sha) + ) + pinned = current if same_git_repo else resolve_repo(str(scope["root"])) + return RepoContext( + cwd=current.cwd, + root=pinned.root, + identity=str(scope["repo_id"]), + app_id=str(scope["app_id"]), + branch=pinned.branch, + head_sha=pinned.head_sha, + project_id=pinned.project_id, + directory=str(scope["directory"] or ""), + ) + + def prepare_flush( + self, repo: RepoContext, session_id: str, reason: str + ) -> tuple[str, list[dict[str, Any]]] | None: + with self.conn: + self.conn.execute("BEGIN IMMEDIATE") + existing = self.conn.execute( + """SELECT * FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status NOT IN ('semantic-succeeded', 'explicitly-stored', 'gave-up') + ORDER BY created_at LIMIT 1""", + (repo.identity, session_id), + ).fetchone() + if existing and int(existing["attempts"] or 0) >= MAX_FLUSH_ATTEMPTS: + self.conn.execute( + "UPDATE flushes SET status = 'gave-up', updated_at = ? WHERE packet_id = ?", + (utc_now(), existing["packet_id"]), + ) + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status="gave-up", + success=False, + attempts=int(existing["attempts"] or 0), + ) + existing = None + if existing: + if reason != "periodic" and existing["reason"] == "periodic": + self.conn.execute( + "UPDATE flushes SET reason = ?, updated_at = ? WHERE packet_id = ?", + (reason, utc_now(), existing["packet_id"]), + ) + existing_rows = self.conn.execute( + "SELECT * FROM events WHERE flush_id = ? ORDER BY id", + (existing["packet_id"],), + ).fetchall() + if existing_rows: + return str(existing["packet_id"]), [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in existing_rows + ] + + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo.identity, session_id), + ).fetchall() + if not rows: + return None + + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + events = select_checkpoint_events(events, force=reason != "periodic") + if not events: + return None + event_start, event_end = events[0]["id"], events[-1]["id"] + packet_material = f"{repo.identity}\0{session_id}\0{event_start}\0{event_end}" + packet_id = hashlib.sha256(packet_material.encode()).hexdigest()[:32] + now = utc_now() + + self.conn.execute( + """INSERT OR IGNORE INTO flushes + (packet_id, repo_id, app_id, session_id, reason, event_start, + event_end, status, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, 'prepared', ?, ?)""", + ( + packet_id, + repo.identity, + repo.app_id, + session_id, + reason, + event_start, + event_end, + now, + now, + ), + ) + event_ids = [event["id"] for event in events] + placeholders = ", ".join("?" for _ in event_ids) + self.conn.execute( + f"UPDATE events SET flush_id = ? " + f"WHERE id IN ({placeholders}) AND flush_id IS NULL", + (packet_id, *event_ids), + ) + return packet_id, events + + def checkpoint_due(self, repo_id: str, session_id: str) -> bool: + if self.has_inflight_flush(repo_id, session_id): + return False + rows = self.conn.execute( + """SELECT * FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id""", + (repo_id, session_id), + ).fetchall() + events = [ + { + "id": row["id"], + "created_at": row["created_at"], + "kind": row["kind"], + "payload": json.loads(row["payload_json"]), + } + for row in rows + ] + return bool(select_checkpoint_events(events, force=False)) + + def has_inflight_flush(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM flushes + WHERE repo_id = ? AND session_id = ? + AND status IN ('prepared', 'semantic-queued') + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def flush_record(self, packet_id: str) -> dict[str, Any] | None: + row = self.conn.execute( + "SELECT * FROM flushes WHERE packet_id = ?", (packet_id,) + ).fetchone() + return dict(row) if row else None + + def has_unflushed_events(self, repo_id: str, session_id: str) -> bool: + return ( + self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + LIMIT 1""", + (repo_id, session_id), + ).fetchone() + is not None + ) + + def unflushed_starts_with_session_start( + self, repo_id: str, session_id: str + ) -> bool: + row = self.conn.execute( + """SELECT kind FROM events + WHERE repo_id = ? AND session_id = ? AND flush_id IS NULL + ORDER BY id LIMIT 1""", + (repo_id, session_id), + ).fetchone() + return bool(row and row["kind"] == "session_start") + + def update_flush(self, packet_id: str, **fields: Any) -> None: + allowed = {"status", "episode_event_id", "semantic_event_id", "error"} + updates = {key: value for key, value in fields.items() if key in allowed} + updates["updated_at"] = utc_now() + clause = ", ".join(f"{key} = ?" for key in updates) + failed = str(fields.get("status", "")) in { + "error", + "semantic-failed", + "semantic-timeout", + "semantic-missing", + } + if failed: + clause += ", attempts = attempts + 1" + with self.conn: + self.conn.execute( + f"UPDATE flushes SET {clause} WHERE packet_id = ?", + [*updates.values(), packet_id], + ) + + def unseen( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> list[dict[str, Any]]: + seen = { + row["memory_id"] + for row in self.conn.execute( + "SELECT memory_id FROM retrievals WHERE session_id = ? AND repo_id = ?", + (session_id, repo_id), + ) + } + return [memory for memory in memories if str(memory.get("id", "")) not in seen] + + def mark_injected( + self, session_id: str, repo_id: str, memories: Iterable[dict[str, Any]] + ) -> None: + now = utc_now() + with self.conn: + for rank, memory in enumerate(memories, start=1): + memory_id = str(memory.get("id", "")) + if memory_id: + memory_text = bounded( + memory.get("memory") or memory.get("text") or "", + 4000, + ) + try: + score = float(memory["score"]) + except (KeyError, TypeError, ValueError): + score = None + self.conn.execute( + """INSERT OR IGNORE INTO retrievals + (session_id, repo_id, memory_id, injected_at, rank, + score, memory_text, context_chars) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)""", + ( + session_id, + repo_id, + memory_id, + now, + rank, + score, + memory_text, + len(memory_text), + ), + ) + + def injected_memories( + self, session_id: str, repo_id: str + ) -> list[dict[str, Any]]: + """Return the exact memories already supplied to the main conversation.""" + rows = self.conn.execute( + """SELECT memory_id, rank, score, memory_text + FROM retrievals + WHERE session_id = ? AND repo_id = ? + ORDER BY COALESCE(rank, 2147483647), injected_at, memory_id""", + (session_id, repo_id), + ).fetchall() + return [ + { + "id": row["memory_id"], + "memory": row["memory_text"], + "score": row["score"], + } + for row in rows + if row["memory_text"] + ] + + def start_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + context_chars: int, + ) -> bool: + """Record one native sidekick instance and whether context was first sent.""" + with self.conn: + cursor = self.conn.execute( + """INSERT OR IGNORE INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + context_chars) + VALUES (?, ?, ?, ?, ?, ?)""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + utc_now(), + context_chars, + ), + ) + return int(cursor.rowcount) > 0 + + def stop_sidekick( + self, + repo: RepoContext, + session_id: str, + agent_id: str, + agent_type: str, + transcript_path: str, + final_message: str, + ) -> str: + now = utc_now() + self.conn.execute("BEGIN IMMEDIATE") + try: + if not agent_id: + rows = self.conn.execute( + """SELECT agent_id FROM sidekick_runs + WHERE repo_id = ? AND session_id = ? AND agent_type = ? AND stopped_at IS NULL + LIMIT 2""", + (repo.identity, session_id, agent_type), + ).fetchall() + # Without a host ID, overlapping runs cannot be correlated reliably. + agent_id = rows[0]["agent_id"] if len(rows) == 1 else f"unknown-agent-{time.time_ns()}" + self.conn.execute( + """INSERT INTO sidekick_runs + (repo_id, session_id, agent_id, agent_type, started_at, + stopped_at, transcript_path, final_message) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(repo_id, session_id, agent_id) DO UPDATE SET + stopped_at = excluded.stopped_at, + transcript_path = excluded.transcript_path, + final_message = excluded.final_message""", + ( + repo.identity, + session_id, + agent_id, + agent_type, + now, + now, + bounded(transcript_path, 2000), + redact(final_message).strip(), + ), + ) + self.conn.commit() + except Exception: + self.conn.rollback() + raise + return agent_id + + def operation( + self, + repo: RepoContext, + session_id: str, + operation: str, + duration_ms: float, + success: bool, + *, + item_count: int = 0, + request_chars: int = 0, + response_chars: int = 0, + error: str = "", + ) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO operations + (created_at, repo_id, session_id, operation, duration_ms, + success, item_count, request_chars, response_chars, error) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", + ( + utc_now(), + repo.identity, + session_id, + operation, + duration_ms, + int(success), + item_count, + request_chars, + response_chars, + bounded(error, 1000), + ), + ) + + def has_operation(self, repo_id: str, session_id: str, operation: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM operations + WHERE repo_id = ? AND session_id = ? AND operation = ? + LIMIT 1""", + (repo_id, session_id, operation), + ).fetchone() + return row is not None + + def has_event(self, repo_id: str, session_id: str, kind: str) -> bool: + row = self.conn.execute( + """SELECT 1 FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + return row is not None + + def latest_event_payload( + self, repo_id: str, session_id: str, kind: str + ) -> dict[str, Any]: + row = self.conn.execute( + """SELECT payload_json FROM events + WHERE repo_id = ? AND session_id = ? AND kind = ? + ORDER BY id DESC LIMIT 1""", + (repo_id, session_id, kind), + ).fetchone() + if not row: + return {} + try: + payload = json.loads(row["payload_json"]) + except json.JSONDecodeError: + return {} + return payload if isinstance(payload, dict) else {} + + def setting(self, key: str, default: str = "") -> str: + row = self.conn.execute( + "SELECT value FROM settings WHERE key = ?", (key,) + ).fetchone() + return str(row["value"]) if row else default + + def set_setting(self, key: str, value: str) -> None: + with self.conn: + self.conn.execute( + """INSERT INTO settings(key, value, updated_at) + VALUES (?, ?, ?) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at""", + (key, value, utc_now()), + ) + + def is_paused(self) -> bool: + return self.setting("paused", "false").lower() in { + "1", + "true", + "yes", + "on", + } + + def forget_local_repo(self, repo_id: str) -> dict[str, int]: + tables = { + "events": "repo_id", + "session_scopes": "repo_id", + "flushes": "repo_id", + "retrievals": "repo_id", + "operations": "repo_id", + "sidekick_runs": "repo_id", + } + removed: dict[str, int] = {} + with self.conn: + for table, column in tables.items(): + cursor = self.conn.execute( + f"DELETE FROM {table} WHERE {column} = ?", (repo_id,) + ) + removed[table] = max(int(cursor.rowcount), 0) + return removed + + def status(self, repo_id: str) -> dict[str, Any]: + def count(table: str) -> int: + return int( + self.conn.execute( + f"SELECT COUNT(*) FROM {table} WHERE repo_id = ?", (repo_id,) + ).fetchone()[0] + ) + + last_operation = self.conn.execute( + """SELECT created_at, operation, duration_ms, success, item_count, error + FROM operations WHERE repo_id = ? ORDER BY id DESC LIMIT 1""", + (repo_id,), + ).fetchone() + last_sidekick = self.conn.execute( + """SELECT session_id, agent_id, agent_type, started_at, stopped_at, + context_chars + FROM sidekick_runs WHERE repo_id = ? + ORDER BY started_at DESC LIMIT 1""", + (repo_id,), + ).fetchone() + return { + "paused": self.is_paused(), + "events": count("events"), + "flushes": count("flushes"), + "retrievals": count("retrievals"), + "sidekick_runs": count("sidekick_runs"), + "last_operation": dict(last_operation) if last_operation else None, + "last_sidekick": dict(last_sidekick) if last_sidekick else None, + } + + +def _session_id(hook_input: dict[str, Any]) -> str: + return str(hook_input.get("session_id") or "unknown-session") + + +def record_session_start(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + store.record_event( + repo, + session_id, + "session_start", + { + "source": hook_input.get("source", "startup"), + "model": bounded(hook_input.get("model", ""), 200), + "branch": repo.branch, + "head_sha": repo.head_sha, + }, + ) + telemetry.record( + "session_start", + repo=repo, + session_id=session_id, + trigger=bounded(str(hook_input.get("source", "startup")), 60), + model=bounded(hook_input.get("model", ""), 200), + api_key_configured=bool(api_key()), + is_git_repo=not repo.identity.startswith("local:"), + ) + + +def record_user_prompt( + store: EvidenceStore, hook_input: dict[str, Any] +) -> tuple[RepoContext, str, str, bool]: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prompt = redact(hook_input.get("prompt", "")).strip() + is_first_prompt = not store.has_event(repo.identity, session_id, "user_prompt") + store.record_event(repo, session_id, "user_prompt", {"text": prompt}) + return repo, session_id, prompt, is_first_prompt + + +def _tool_result_preview(response: Any) -> str: + if isinstance(response, dict): + selected = {} + for key in ( + "stdout", + "stderr", + "output", + "content", + "error", + "filePath", + "success", + "interrupted", + ): + if key in response: + selected[key] = response[key] + response = selected or {"keys": sorted(response.keys())[:20]} + return bounded(response, MAX_RESULT_CHARS) + + +def tool_payload(hook_input: dict[str, Any], *, failed: bool | None = False) -> dict[str, Any]: + name = str(hook_input.get("tool_name") or "unknown") + tool_input = hook_input.get("tool_input") or {} + if not isinstance(tool_input, dict): + tool_input = {} + payload: dict[str, Any] = { + "tool": name, + "failed": failed, + "duration_ms": hook_input.get("duration_ms"), + "agent_role": "sidekick" if hook_input.get("agent_id") else "main", + } + if hook_input.get("agent_id"): + payload["agent_id"] = bounded(hook_input["agent_id"], 200) + if hook_input.get("agent_type"): + payload["agent_type"] = bounded(hook_input["agent_type"], 200) + + if name in {"Read", "Write", "Edit", "MultiEdit", "NotebookEdit"}: + path = tool_input.get("file_path") or tool_input.get("notebook_path") + if path: + payload["path"] = bounded(path, 1000) + if name in {"Write", "Edit", "MultiEdit", "NotebookEdit"}: + payload["mutation_chars"] = sum( + len(str(tool_input.get(key, ""))) + for key in ("content", "new_string", "new_source", "edits") + ) + elif name == "Bash" or "command" in tool_input: + command = bounded(tool_input.get("command", ""), MAX_COMMAND_CHARS) + payload["command"] = command + payload["command_kind"] = ( + "test" + if TEST_COMMAND_RE.search(command) + else "build" + if BUILD_COMMAND_RE.search(command) + else "shell" + ) + response = ( + hook_input.get("error") if failed else hook_input.get("tool_response") + ) + payload["result_preview"] = _tool_result_preview(response) + elif name in {"Grep", "Glob", "WebSearch", "WebFetch"}: + for key in ("pattern", "path", "query", "url"): + if tool_input.get(key): + payload[key] = bounded(tool_input[key], 1000) + else: + payload["input_keys"] = sorted(tool_input.keys())[:20] + if failed: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + + if failed and "error" not in payload: + payload["error"] = bounded(hook_input.get("error", ""), MAX_RESULT_CHARS) + return payload + + +def record_tool( + store: EvidenceStore, hook_input: dict[str, Any], *, failed: bool | None = False +) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + payload = tool_payload(hook_input, failed=failed) + if payload.get("path"): + payload["repo_path"] = _repo_relative_path(repo, str(payload["path"])) + store.record_event( + repo, + session_id, + "tool_failure" if failed else "tool_result", + payload, + ) + + +def record_sidekick_start( + store: EvidenceStore, hook_input: dict[str, Any], *, inject_context: bool = True +) -> str: + """Record a native sidekick and reuse the main turn's retrieved memories.""" + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_id = bounded(hook_input.get("agent_id", "unknown-agent"), 200) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + context = combine_context( + format_context(store.injected_memories(session_id, repo.identity)) + ) + if not inject_context: + context = "" + first_start = store.start_sidekick( + repo, session_id, agent_id, agent_type, len(context) + ) + store.record_event( + repo, + session_id, + "sidekick_start", + { + "agent_id": agent_id, + "agent_type": agent_type, + "context_chars": len(context) if first_start else 0, + "worktree_root": bounded(repo.root, 2000), + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="start", + first_start=first_start, + context_chars=len(context) if first_start else 0, + ) + return context if first_start else "" + + +def record_sidekick_stop(store: EvidenceStore, hook_input: dict[str, Any]) -> None: + session_id = _session_id(hook_input) + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + agent_type = bounded(hook_input.get("agent_type", "mem0:sidekick"), 200) + agent_id = bounded(hook_input.get("agent_id", ""), 200) + final_message = redact(hook_input.get("last_assistant_message", "")).strip() + transcript_path = bounded(hook_input.get("agent_transcript_path", ""), 2000) + agent_id = store.stop_sidekick( + repo, + session_id, + agent_id, + agent_type, + transcript_path, + final_message, + ) + store.record_event( + repo, + session_id, + "sidekick_stop", + { + "agent_id": agent_id, + "agent_type": agent_type, + "transcript_path": transcript_path, + "final_message": final_message, + }, + ) + telemetry.record( + "sidekick", + repo=repo, + session_id=session_id, + phase="stop", + has_transcript=bool(transcript_path), + message_chars=len(final_message), + ) + + +def _ordered_unique(values: Iterable[str]) -> list[str]: + seen: set[str] = set() + result = [] + for value in values: + if value and value not in seen: + seen.add(value) + result.append(value) + return result + + +def _repo_relative_path(repo: RepoContext, value: str) -> str: + value = str(value or "").strip() + if not value: + return "" + try: + path = Path(value) + if path.is_absolute(): + try: + return path.resolve().relative_to(Path(repo.root).resolve()).as_posix() + except ValueError: + return "" + except (OSError, ValueError): + pass + return bounded(value, 1000) + + +def _render_command_lines(commands: list[dict[str, str]]) -> list[str]: + lines = [] + for command in commands: + line = f"- [{command['status']}/{command['kind']}] {command['command']}" + if command["result"]: + line += f" — {bounded(command['result'], 500).replace(chr(10), ' ')}" + lines.append(line) + return lines + + +def build_episode( + repo: RepoContext, + session_id: str, + packet_id: str, + events: list[dict[str, Any]], + *, + canonical_task: str = "", + task_outcome: str = "", +) -> tuple[str, dict[str, Any]]: + prompts = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "user_prompt" and e["payload"].get("text") + ] + assistant_conclusions = [ + redact(e["payload"].get("text", "")).strip() + for e in events + if e["kind"] == "assistant_stop" and e["payload"].get("text") + ] + sidekick_outcomes = [ + redact(e["payload"].get("final_message", "")).strip() + for e in events + if e["kind"] == "sidekick_stop" and e["payload"].get("final_message") + ] + tools = [ + e["payload"] for e in events if e["kind"] in {"tool_result", "tool_failure"} + and e["payload"].get("agent_role", "main") == "main" + ] + read_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") == "Read" + ) + modified_paths = _ordered_unique( + _repo_relative_path(repo, str(t.get("repo_path") or t.get("path", ""))) + for t in tools + if t.get("tool") in {"Write", "Edit", "MultiEdit", "NotebookEdit"} + ) + searches = [ + {key: t[key] for key in ("tool", "pattern", "path", "query", "url") if key in t} + for t in tools + if t.get("tool") in {"Grep", "Glob", "WebSearch", "WebFetch"} + ] + commands = [ + { + "command": t.get("command", ""), + "kind": t.get("command_kind", "shell"), + "status": "unknown" if t.get("failed", False) is None else "failed" if t.get("failed") else "succeeded", + "result": t.get("result_preview", ""), + } + for t in tools + if t.get("command") + ] + + task = bounded(canonical_task or (prompts[0] if prompts else ""), 4000) + conclusion = redact(assistant_conclusions[-1] if assistant_conclusions else "").strip() + outcome = bounded(task_outcome, 2000) + + extraction_messages: list[dict[str, str]] = [] + pending_user_messages: list[dict[str, str]] = [] + if task and not prompts: + pending_user_messages.append({"role": "user", "content": task}) + for event in events: + if event["kind"] == "user_prompt" and event["payload"].get("text"): + pending_user_messages.append( + { + "role": "user", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + elif event["kind"] == "assistant_stop": + transcript_messages = event["payload"].get("transcript_messages") or [] + if isinstance(transcript_messages, list) and transcript_messages: + transcript_users = { + redact(message.get("content") or "").strip() + for message in transcript_messages + if isinstance(message, dict) and message.get("role") == "user" + } + extraction_messages.extend( + message + for message in pending_user_messages + if message["content"].strip() not in transcript_users + ) + extraction_messages.extend( + { + "role": str(message.get("role") or ""), + "content": redact(message.get("content") or "").strip(), + } + for message in transcript_messages + if isinstance(message, dict) + and message.get("role") in {"user", "assistant"} + and message.get("content") + ) + else: + extraction_messages.extend(pending_user_messages) + if event["payload"].get("text"): + extraction_messages.append( + { + "role": "assistant", + "content": redact(event["payload"].get("text", "")).strip(), + } + ) + pending_user_messages = [] + elif event["kind"] == "sidekick_stop": + pass + extraction_messages.extend(pending_user_messages) + + structured = { + "packet_id": packet_id, + "repo": repo.identity, + "app_id": repo.app_id, + "session_id": session_id, + "branch": repo.branch, + "head_sha": repo.head_sha, + "task": task, + "task_outcome": outcome, + "assistant_conclusion": conclusion, + "user_messages": prompts, + "assistant_outcomes": assistant_conclusions, + "sidekick_outcomes": sidekick_outcomes, + "extraction_messages": extraction_messages, + "files_read": read_paths[:50], + "files_modified": modified_paths[:50], + "searches": searches[-30:], + "commands": commands[-30:], + } + lines = ["Coding-session episode"] + if task: + lines.extend(["", "Task:", task]) + if modified_paths: + lines.extend( + ["", "Files modified:", *[f"- {path}" for path in modified_paths[:50]]] + ) + if read_paths: + lines.extend(["", "Files read:", *[f"- {path}" for path in read_paths[:50]]]) + if commands: + lines.append("") + lines.append("Observed commands:") + lines.extend(_render_command_lines(commands[-30:])) + if searches: + lines.extend( + [ + "", + "Observed searches:", + *[ + f"- {json.dumps(item, ensure_ascii=False, sort_keys=True)}" + for item in searches[-20:] + ], + ] + ) + if conclusion: + lines.extend(["", "Agent conclusion:", conclusion]) + if outcome: + lines.extend(["", "Task outcome:", outcome]) + lines.extend( + [ + "", + f"Provenance: repo={repo.identity}; branch={repo.branch}; head={repo.head_sha}; packet={packet_id}", + ] + ) + content = "\n".join(lines) + return bounded(content, MAX_EPISODE_CHARS), structured + + +def build_semantic_evidence(structured: dict[str, Any]) -> str: + """Format changed paths for memory extraction. + + Test and build results remain in the local evidence store for diagnostics, + but are not useful repository knowledge by default and should not steer + memory extraction toward transient verification details. + """ + modified_paths = [ + bounded(path, 500) for path in structured.get("files_modified", [])[:20] + ] + commands = structured.get("commands") or [] + if not any(command.get("status") == "failed" for command in commands): + commands = [] + + if not modified_paths and not commands: + return "" + + lines = ["Additional repository details from this session"] + if modified_paths: + lines.extend( + [ + "", + "Changed paths:", + *[f"- {path}" for path in modified_paths], + ] + ) + if commands: + lines.extend(["", "Commands run in this session:", *_render_command_lines(commands)]) + return bounded("\n".join(lines), 8000) + + +def build_extraction_messages(structured: dict[str, Any]) -> list[dict[str, str]]: + """Build the session messages sent to Mem0 for memory extraction.""" + evidence = build_semantic_evidence(structured) + messages = [ + {"role": message["role"], "content": redact(message["content"]).strip()} + for message in structured.get("extraction_messages", []) + if message.get("role") in {"user", "assistant"} and message.get("content") + ] + if evidence: + for message in reversed(messages): + if message["role"] == "assistant": + message["content"] = f"{message['content']}\n\n{evidence}" + break + else: + messages.append({"role": "assistant", "content": evidence}) + + return messages + + +def _estimated_tokens(value: str) -> int: + """Conservatively estimate tokens without adding a tokenizer dependency.""" + ascii_chars = sum(ord(char) < 128 for char in value) + return math.ceil((ascii_chars * 0.4) + (len(value) - ascii_chars)) + + +def _message_tokens(messages: list[dict[str, str]]) -> int: + return _estimated_tokens(json.dumps(messages, ensure_ascii=False)) + + +def _is_agent_assignment(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent assignment (") + + +def _is_agent_response(message: dict[str, str]) -> bool: + return message.get("role") == "assistant" and message.get( + "content", "" + ).startswith("Subagent response (") + + +def extraction_message_batches( + messages: list[dict[str, str]], + *, + max_tokens: int = MAX_EXTRACTION_INPUT_TOKENS, +) -> list[list[dict[str, str]]]: + """Keep exchanges together when possible; split oversized messages to enforce the request budget.""" + if not messages or _message_tokens(messages) <= max_tokens: + return [messages] + + exchanges: list[list[dict[str, str]]] = [] + exchange: list[dict[str, str]] = [] + for message in messages: + if message.get("role") == "user" and exchange: + exchanges.append(exchange) + exchange = [] + exchange.append(message) + if exchange: + exchanges.append(exchange) + + units: list[list[dict[str, str]]] = [] + for exchange in exchanges: + if _message_tokens(exchange) <= max_tokens: + units.append(exchange) + continue + index = 0 + while index < len(exchange): + message = exchange[index] + if ( + _is_agent_assignment(message) + and index + 1 < len(exchange) + and _is_agent_response(exchange[index + 1]) + ): + units.append(exchange[index : index + 2]) + index += 2 + else: + units.append([message]) + index += 1 + + bounded_units: list[list[dict[str, str]]] = [] + for unit in units: + if _message_tokens(unit) <= max_tokens: + bounded_units.append(unit) + continue + for message in unit: + remaining = message["content"] + while remaining: + low, high = 0, len(remaining) + while low < high: + middle = (low + high + 1) // 2 + if _message_tokens([{**message, "content": remaining[:middle]}]) <= max_tokens: + low = middle + else: + high = middle - 1 + if low == 0: + raise ValueError("Extraction token budget cannot fit a message") + bounded_units.append([{**message, "content": remaining[:low]}]) + remaining = remaining[low:] + + batches: list[list[dict[str, str]]] = [] + batch: list[dict[str, str]] = [] + for unit in bounded_units: + candidate = [*batch, *unit] + if batch and _message_tokens(candidate) > max_tokens: + batches.append(batch) + batch = list(unit) + else: + batch = candidate + if batch: + batches.append(batch) + return batches + + +def _request_json( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + raw = json.dumps(payload, ensure_ascii=False).encode() + request = urllib.request.Request( + url, + data=raw, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + parsed = json.loads(response_raw or b"{}") + return parsed, len(raw), len(response_raw) + + +def _request_json_with_network_retry( + url: str, key: str, payload: dict[str, Any], timeout: float +) -> tuple[dict[str, Any] | list[Any], int, int]: + """Retry one transient connection failure without retrying API responses.""" + try: + return _request_json(url, key, payload, timeout) + except urllib.error.HTTPError: + raise + except (urllib.error.URLError, TimeoutError, OSError): + time.sleep(0.25) + return _request_json(url, key, payload, timeout) + + +def _get_json( + url: str, key: str, timeout: float +) -> tuple[dict[str, Any] | list[Any], int]: + request = urllib.request.Request( + url, + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="GET", + ) + with urllib.request.urlopen(request, timeout=timeout) as response: + response_raw = response.read() + return json.loads(response_raw or b"{}"), len(response_raw) + + +def _event_id(response: dict[str, Any] | list[Any]) -> str: + return str(response.get("event_id", "")) if isinstance(response, dict) else "" + + +def _stored_event_ids(value: Any) -> list[str]: + raw = str(value or "") + if not raw.startswith("["): + return [] + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + return [] + return [str(item or "") for item in parsed] if isinstance(parsed, list) else [] + + +def _result_count(response: dict[str, Any] | list[Any]) -> int: + if isinstance(response, dict): + results = response.get("results") + return len(results) if isinstance(results, list) else 0 + return len(response) if isinstance(response, list) else 0 + + +def touch_handoff_heartbeat() -> None: + """Mark the worker's handoff file alive so recovery does not relaunch it.""" + path = os.environ.get("MEM0_CODE_HANDOFF_PATH", "") + if not path: + return + try: + os.utime(path) + except OSError: + pass + + +def _wait_for_event(api_url: str, key: str, event_id: str) -> tuple[str, int, int]: + """Wait for extraction to finish before a later task can search the store.""" + if not event_id: + return "MISSING", 0, 0 + wait_seconds = float(os.environ.get("MEM0_CODE_EXTRACTION_WAIT_SECONDS", "120")) + poll_seconds = max(float(os.environ.get("MEM0_CODE_EVENT_POLL_SECONDS", "1")), 0.1) + deadline = time.monotonic() + wait_seconds + response_chars = 0 + while time.monotonic() < deadline: + touch_handoff_heartbeat() + try: + response, size = _get_json( + f"{api_url}/v1/event/{event_id}/", + key, + min(10, poll_seconds + 5), + ) + except urllib.error.HTTPError as exc: + if exc.code not in {408, 429} and exc.code < 500: + raise + time.sleep(poll_seconds) + continue + except (urllib.error.URLError, TimeoutError, OSError): + # The extraction job is durable server-side. A transient polling + # failure must not discard a job that may still complete normally. + time.sleep(poll_seconds) + continue + response_chars += size + status = ( + str(response.get("status", "UNKNOWN")) + if isinstance(response, dict) + else "UNKNOWN" + ) + if status in {"SUCCEEDED", "FAILED"}: + return status, response_chars, _result_count(response) + time.sleep(poll_seconds) + return "TIMEOUT", response_chars, 0 + + +def _record_flush( + repo: RepoContext, + session_id: str, + reason: str, + status: str, + elapsed: float, + **extra: Any, +) -> None: + telemetry.record( + "flush", + repo=repo, + session_id=session_id, + reason=reason, + status=status, + success=status in {"semantic-succeeded", "nothing-to-flush"}, + duration_ms=round(elapsed, 2), + **extra, + ) + + +def flush_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + key = api_key() + if not key: + telemetry.record("flush", reason=reason, status="local-only", success=False) + return {"status": "local-only", "reason": "no-api-key"} + + session_id = _session_id(hook_input) + if session_id == "unknown-session": + telemetry.record("flush", reason=reason, status="no-session-id", success=False) + return {"status": "error", "reason": "no-session-id"} + repo = store.repo_for_session(session_id, hook_input.get("cwd")) + prepared = store.prepare_flush(repo, session_id, reason) + if prepared is None: + return {"status": "nothing-to-flush"} + packet_id, events = prepared + existing_flush = store.flush_record(packet_id) or {} + + _, structured = build_episode( + repo, + session_id, + packet_id, + events, + canonical_task=bounded(hook_input.get("task", ""), 4000), + task_outcome=bounded(hook_input.get("task_outcome", ""), 2000), + ) + + metadata = {"source": _harness_source_tag} + if repo.branch and repo.branch not in {"detached", "unknown"}: + metadata["branch"] = repo.branch + if repo.head_sha: + metadata["git_sha"] = repo.head_sha + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + add_url = f"{api_url}/v3/memories/add/" + + write_user = _scope_value(user_id()) + write_project = _scope_value(repo.project_id) + if not write_user or not _scope_value(repo.app_id) or not write_project: + telemetry.record("flush", reason=reason, status="unscoped", success=False) + return {"status": "error", "reason": "wildcard-scope"} + + body = { + "agent_id": write_project, + "user_id": write_user, + "app_id": repo.app_id, + "run_id": session_id, + "metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)}, + "agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS, + "custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS, + "custom_categories": CODING_MEMORY_CATEGORIES, + "infer": True, + } + + started = time.perf_counter() + try: + stored_events = _stored_event_ids(existing_flush.get("semantic_event_id")) + existing_event = ( + "" if stored_events else str(existing_flush.get("semantic_event_id") or "") + ) + if existing_event: + existing_status, existing_resp, existing_items = _wait_for_event( + api_url, key, existing_event + ) + if existing_status == "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=existing_event, + error="", + ) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + True, + item_count=existing_items, + response_chars=existing_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=existing_items, + resumed=True, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "semantic_status": existing_status, + "memory_count": existing_items, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + if existing_status == "TIMEOUT": + elapsed = (time.perf_counter() - started) * 1000 + error = "semantic extraction event timed out" + store.update_flush(packet_id, status="semantic-timeout", error=error) + store.operation( + repo, + session_id, + "flush-retry", + elapsed, + False, + response_chars=existing_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-timeout", + elapsed, + resumed=True, + error_kind="timeout", + ) + return { + "status": "semantic-timeout", + "packet_id": packet_id, + "semantic_event_id": existing_event, + "duration_ms": round(elapsed, 2), + "resumed": True, + } + + message_batches = [ + batch + for batch in extraction_message_batches( + build_extraction_messages(structured) + ) + if batch + ] + batches = [(body, messages) for messages in message_batches] + if not batches: + store.update_flush(packet_id, status="semantic-succeeded", error="") + return {"status": "nothing-to-flush", "packet_id": packet_id} + operation_name = "flush-retry" if stored_events else "flush" + semantic_events = stored_events[: len(batches)] + semantic_events += [""] * (len(batches) - len(semantic_events)) + semantic_req = 0 + semantic_resp = 0 + for index, (body, messages) in enumerate(batches): + if semantic_events[index]: + continue + semantic_response, request_chars, response_chars = _request_json( + add_url, + key, + {**body, "messages": messages}, + 15, + ) + semantic_events[index] = _event_id(semantic_response) + semantic_req += request_chars + semantic_resp += response_chars + store.update_flush( + packet_id, + status="semantic-queued", + semantic_event_id=json.dumps(semantic_events), + ) + + semantic_event = semantic_events[-1] + semantic_status = "SUCCEEDED" + event_resp = 0 + semantic_items = 0 + failed_event = semantic_event + for index, queued_event in enumerate(semantic_events): + status, response_chars, item_count = _wait_for_event( + api_url, key, queued_event + ) + event_resp += response_chars + semantic_items += item_count + if status != "SUCCEEDED": + semantic_status = status + failed_event = queued_event + if status in {"FAILED", "MISSING"}: + semantic_events[index] = "" + store.update_flush( + packet_id, semantic_event_id=json.dumps(semantic_events) + ) + break + if semantic_status != "SUCCEEDED": + elapsed = (time.perf_counter() - started) * 1000 + error = f"semantic extraction event {semantic_status.lower()}" + store.update_flush( + packet_id, status=f"semantic-{semantic_status.lower()}", error=error + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + False, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + error=error, + ) + _record_flush( + repo, + session_id, + reason, + f"semantic-{semantic_status.lower()}", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + error_kind=telemetry.error_kind(error), + ) + return { + "status": f"semantic-{semantic_status.lower()}", + "packet_id": packet_id, + "semantic_event_id": failed_event, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + + elapsed = (time.perf_counter() - started) * 1000 + store.update_flush( + packet_id, + status="semantic-succeeded", + semantic_event_id=json.dumps(semantic_events), + error="", + ) + store.operation( + repo, + session_id, + operation_name, + elapsed, + True, + item_count=semantic_items, + request_chars=semantic_req, + response_chars=semantic_resp + event_resp, + ) + _record_flush( + repo, + session_id, + reason, + "semantic-succeeded", + elapsed, + memory_count=semantic_items, + batch_count=len(batches), + request_chars=semantic_req, + ) + effective_reason = str( + (store.flush_record(packet_id) or {}).get("reason") or reason + ) + if ( + store.has_unflushed_events(repo.identity, session_id) + and not store.unflushed_starts_with_session_start( + repo.identity, session_id + ) + and ( + effective_reason != "periodic" + or store.checkpoint_due(repo.identity, session_id) + ) + ): + return flush_session(store, hook_input, effective_reason) + return { + "status": "semantic-succeeded", + "packet_id": packet_id, + "semantic_event_id": semantic_event, + "semantic_status": semantic_status, + "memory_count": semantic_items, + "duration_ms": round(elapsed, 2), + } + except Exception as exc: # hooks must fail open + elapsed = (time.perf_counter() - started) * 1000 + error = bounded(str(exc), 1000) + store.update_flush(packet_id, status="error", error=error) + store.operation(repo, session_id, "flush", elapsed, False, error=error) + _record_flush( + repo, + session_id, + reason, + "error", + elapsed, + error_kind=telemetry.error_kind(exc), + ) + return {"status": "error", "packet_id": packet_id, "error": error} + + +def checkpoint_session( + store: EvidenceStore, hook_input: dict[str, Any], reason: str +) -> dict[str, Any]: + """Run remote extraction at a durable boundary.""" + return flush_session(store, hook_input, reason) + + + +def search_memories( + store: EvidenceStore | None, + repo: RepoContext, + session_id: str | None, + query: str, + *, + top_k: int | None = None, + category: str | None = None, + scope: str | None = None, + run_id: str | None = None, + operation: str = "search", + timeout: float = 5, +) -> MemorySearchResult: + key = api_key() + if not key or not query.strip(): + return MemorySearchResult(False, 0, 0, []) + search_once = os.environ.get( + "MEM0_CODE_SEARCH_ONCE_PER_SESSION", "false" + ).lower() in { + "1", + "true", + "yes", + "on", + } + track_session = store is not None and bool(session_id) + if ( + search_once + and track_session + and store.has_operation(repo.identity, session_id, "search") + ): + return MemorySearchResult(False, 0, 0, []) + + result_limit = min( + max( + top_k + if top_k is not None + else _int_option("top_k", "MEM0_CODE_TOP_K", 3), + 1, + ), + 20, + ) + if category is not None and category not in CODING_MEMORY_CATEGORY_NAMES: + raise ValueError(f"Unknown memory category: {category}") + user, project = _scope_value(user_id()), _scope_value(repo.project_id) + if not user or not project or not _scope_value(repo.app_id): + return MemorySearchResult(False, 0, 0, []) + filters = _search_filters(user, repo, resolve_search_scope(scope)) + if category: + filters = {"AND": [filters, {"categories": {"contains": category}}]} + if run_id: + filters = {"AND": [filters, {"run_id": run_id}]} + payload = { + "query": query, + "app_id": repo.app_id, + "filters": filters, + "top_k": result_limit, + "rerank": False, + "latest_only": True, + } + url = ( + os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + + "/v3/memories/search/" + ) + started = time.perf_counter() + try: + response, request_chars, response_chars = _request_json_with_network_retry( + url, key, payload, timeout + ) + memories = ( + response if isinstance(response, list) else response.get("results", []) + ) + memories = [ + memory + for memory in memories + if isinstance(memory, dict) + and (memory.get("metadata") or {}).get("record_kind") != "task_episode" + ][:result_limit] + if track_session: + returned_memories = store.unseen(session_id, repo.identity, memories) + store.mark_injected(session_id, repo.identity, returned_memories) + already_shown_count = len(memories) - len(returned_memories) + else: + returned_memories = memories + already_shown_count = 0 + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation( + repo, + session_id, + operation, + elapsed, + True, + item_count=len(returned_memories), + request_chars=request_chars, + response_chars=response_chars, + ) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=True, + duration_ms=round(elapsed, 2), + matched_count=len(memories), + returned_count=len(returned_memories), + already_shown_count=already_shown_count, + top_k=result_limit, + has_category=bool(category), + ) + return MemorySearchResult( + succeeded=True, + matched_count=len(memories), + already_shown_count=already_shown_count, + memories=returned_memories, + ) + except Exception as exc: + elapsed = (time.perf_counter() - started) * 1000 + if track_session: + store.operation(repo, session_id, operation, elapsed, False, error=str(exc)) + telemetry.record( + "search", + repo=repo, + session_id=session_id, + trigger=operation, + success=False, + duration_ms=round(elapsed, 2), + top_k=result_limit, + has_category=bool(category), + error_kind=telemetry.error_kind(exc), + ) + return MemorySearchResult(False, 0, 0, []) + + +def format_context( + memories: list[dict[str, Any]], + heading: str = "Relevant repository memories:", +) -> str: + if not memories: + return "" + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + lines = [heading] if heading else [] + for memory in memories: + text = re.sub( + r"\s+", + " ", + redact(memory.get("memory") or memory.get("text") or ""), + ) + text = text.strip() + if not text: + continue + branch = str((memory.get("metadata") or {}).get("branch") or "").strip() + branch_label = ( + f" [learnt on branch {branch}]" + if branch.casefold() not in {"", "main", "master", "unknown", "detached"} + else "" + ) + number = len(lines) if heading else len(lines) + 1 + entry = f"{number}. {text}{branch_label}" + candidate = "\n".join([*lines, entry]) + if len(candidate) <= limit: + lines.append(entry) + continue + if not lines or (heading and len(lines) == 1): + prefix = f"{number}. " + suffix = f"…{branch_label}" + available = ( + limit + - len("\n".join(lines)) + - (1 if lines else 0) + - len(prefix) + - len(suffix) + ) + if available > 0: + lines.append(prefix + text[:available].rstrip() + suffix) + break + minimum_lines = 2 if heading else 1 + return "\n".join(lines) if len(lines) >= minimum_lines else "" + + +def format_search_result(result: MemorySearchResult) -> str: + """Return only the text the coding agent needs from an explicit memory search.""" + if not result.succeeded: + return "Memory search failed." + if result.memories: + rendered = format_context(result.memories, heading="") + if rendered: + return rendered + return "No matching memories found." + + +def combine_context(*contexts: str) -> str: + """Combine memory sources under one hard budget without repeated lines.""" + seen: set[str] = set() + lines: list[str] = [] + for context in contexts: + for line in str(context or "").splitlines(): + normalized = re.sub(r"\s+", " ", line).strip().casefold() + if not normalized or normalized in seen: + continue + seen.add(normalized) + lines.append(line.rstrip()) + limit = min( + max( + _int_option( + "max_context_chars", + "MEM0_CODE_MAX_CONTEXT_CHARS", + DEFAULT_MAX_CONTEXT_CHARS, + ), + 1000, + ), + 10000, + ) + return bounded("\n".join(lines), limit) if lines else "" + + +def _scoped_memory_ids( + api_url: str, key: str, user: str, repo: RepoContext, include_project: bool +) -> list[str]: + """List this user's memory ids for this repository, plus the shared project memory when asked.""" + ids: list[str] = [] + seen: set[str] = set() + prefix = repo.app_id + _collect_memory_ids( + api_url, key, {"user_id": user}, ids, seen, + app_id_prefix=prefix, + ) + if include_project: + for project_id in _shared_project_ids(repo): + _collect_memory_ids( + api_url, key, {"agent_id": project_id}, ids, seen, + app_id_prefix=prefix, + ) + return ids + + +def _collect_memory_ids( + api_url: str, + key: str, + filters: dict[str, Any], + ids: list[str], + seen: set[str], + *, + app_id_prefix: str = "", +) -> None: + """Page through one list filter; the list endpoint returns nothing for an OR whose user branch has no memories.""" + payload = {"filters": filters} + for page in range(1, FORGET_MAX_PAGES + 1): + parsed, _, _ = _request_json( + f"{api_url}/v2/memories/?page={page}&page_size={FORGET_PAGE_SIZE}", + key, + payload, + 15, + ) + items = parsed.get("results") if isinstance(parsed, dict) else parsed + if not isinstance(items, list) or not items: + break + for item in items: + if not isinstance(item, dict): + continue + memory_id = str(item.get("id", "")) + if not memory_id or memory_id in seen: + continue + if app_id_prefix: + item_app_id = str(item.get("app_id") or "") + if item_app_id != app_id_prefix and not item_app_id.startswith(app_id_prefix + "/"): + continue + seen.add(memory_id) + ids.append(memory_id) + if len(items) < FORGET_PAGE_SIZE: + break + + +def _delete_memory(api_url: str, key: str, memory_id: str) -> bool: + request = urllib.request.Request( + f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/", + headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}, + method="DELETE", + ) + try: + with urllib.request.urlopen(request, timeout=15): + return True + except Exception: + return False + + +def forget_remote_repo( + repo: RepoContext, *, include_project_memory: bool = False +) -> dict[str, Any]: + """Delete this user's memories for this repository; project memory is shared, so only on request.""" + key = api_key() + if not key: + telemetry.record("forget", repo=repo, success=False, error_kind="no-api-key") + return {"status": "error", "error": "Mem0 API key is not configured"} + user = _scope_value(user_id()) + if not user or not _scope_value(repo.app_id) or not _scope_value(repo.project_id): + telemetry.record("forget", repo=repo, success=False, error_kind="unscoped") + return { + "status": "error", + "error": "Refusing to forget: the user or repository scope is a wildcard", + } + api_url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + try: + memory_ids = _scoped_memory_ids(api_url, key, user, repo, include_project_memory) + except Exception as exc: + telemetry.record( + "forget", repo=repo, success=False, error_kind=telemetry.error_kind(exc) + ) + return {"status": "error", "error": bounded(str(exc), 1000)} + deleted = sum(_delete_memory(api_url, key, memory_id) for memory_id in memory_ids) + failed = len(memory_ids) - deleted + telemetry.record("forget", repo=repo, success=not failed, item_count=deleted) + if failed: + return { + "status": "partial", + "deleted": deleted, + "failed": failed, + "error": f"{failed} of {len(memory_ids)} memories could not be deleted", + } + return {"status": "deleted", "deleted": deleted} + + +def _doctor_mem0_authentication(repo: RepoContext) -> dict[str, Any]: + """Verify the configured key with one read-only, repository-scoped search.""" + key = api_key() + if not key: + return {"ok": False, "detail": "API key missing"} + payload = { + "query": "Mem0 authentication check", + "filters": { + "AND": [ + {"user_id": user_id()}, + {"app_id": repo.app_id}, + ] + }, + "top_k": 1, + "threshold": 1.0, + "rerank": False, + } + url = os.environ.get("MEM0_API_URL", DEFAULT_API_URL).rstrip("/") + started = time.perf_counter() + try: + _request_json(f"{url}/v3/memories/search/", key, payload, 5) + except Exception as exc: + return {"ok": False, "detail": bounded(str(exc), 300)} + elapsed = (time.perf_counter() - started) * 1000 + return {"ok": True, "detail": f"connected ({elapsed:.0f} ms)"} + + +def _doctor_user_id() -> dict[str, Any]: + """Flag a configured user ID the plugin refuses, since the silent fallback surprises people.""" + configured = _plugin_option("user_id", "MEM0_CODE_USER_ID") or os.environ.get( + "MEM0_USER_ID", "" + ) + if configured and not _scope_value(configured): + return { + "ok": False, + "detail": f"configured user_id {configured!r} is a wildcard; using {user_id()!r}", + } + return {"ok": True, "detail": user_id()} + + +def doctor(cwd: str | None = None) -> dict[str, Any]: + repo = resolve_repo(cwd) + directory = data_dir() + directory.mkdir(parents=True, exist_ok=True) + checks: dict[str, dict[str, Any]] = { + "python": { + "ok": tuple(sys.version_info[:2]) >= (3, 10), + "detail": f"{sys.version_info.major}.{sys.version_info.minor}", + }, + "data_directory": { + "ok": os.access(directory, os.W_OK), + "detail": str(directory), + }, + "mem0_api_key": { + "ok": bool(api_key()), + "detail": "configured" if api_key() else "missing", + }, + "repository": { + "ok": bool(repo.identity), + "detail": repo.identity, + }, + "user_id": _doctor_user_id(), + "mem0_authentication": _doctor_mem0_authentication(repo), + } + return { + "ok": all(bool(value["ok"]) for value in checks.values()), + "plugin_version": PLUGIN_VERSION, + "repo_id": repo.identity, + "app_id": repo.app_id, + "user_id": user_id(), + "checks": checks, + } diff --git a/integrations/mem0-agent-plugin/core/telemetry.py b/integrations/mem0-agent-plugin/core/telemetry.py new file mode 100644 index 000000000..249595475 --- /dev/null +++ b/integrations/mem0-agent-plugin/core/telemetry.py @@ -0,0 +1,397 @@ +#!/usr/bin/env python3 +"""Anonymous usage telemetry for Mem0 agent plugins. + +Hooks run on a 3-6 second budget and fire on every tool call, so recording never +touches the network: `record` appends one JSON line to a local spool and returns. +A detached `python3 telemetry.py` drains the spool in one batched PostHog request, +started once per session and again from the flush worker that is already detached. + +Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false. + +Never sends prompts, memory text, queries, file paths, repository names, or API +keys: only event names, durations, counts, coarse outcomes, and salted hashes. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import platform +import subprocess +import sys +import time +import urllib.error +import urllib.request +import uuid +from pathlib import Path +from typing import Any + +import memory_core + +_harness: str = "generic" +_source_tag: str = "MEM0_PLUGIN" +_PRIVATE_KEYS = { + "apikey", + "authorization", + "password", + "query", + "secret", + "prompt", + "token", + "text", + "memory", + "message", + "error", + "path", + "cwd", + "userid", + "agentid", + "runid", + "repoid", + "repositoryid", + "projectid", + "appid", + "filters", +} + + +def init(harness: str = "generic", source_tag: str = "") -> None: + global _harness, _source_tag + _harness = harness + _source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN" + +POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" +POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" +POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/" +EVENT_PREFIX = "code" +SPOOL_LIMIT_BYTES = 256 * 1024 +BATCH_SIZE = 100 +SEND_TIMEOUT = 5 +CLAIM_STALE_SECONDS = 120 +CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60 + + +def is_enabled() -> bool: + """Whether telemetry is switched on for this process.""" + return os.environ.get("MEM0_TELEMETRY", "true").strip().lower() not in { + "false", + "0", + "no", + "off", + } + + +def _digest(value: str, length: int = 16) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length] + + +def _safe_value(value: Any) -> Any: + if isinstance(value, str): + return memory_core.redact(value) + if isinstance(value, dict): + return { + key: _safe_value(item) + for key, item in value.items() + if "".join(character for character in str(key).lower() if character.isalnum()) + not in _PRIVATE_KEYS + } + if isinstance(value, (list, tuple)): + return [_safe_value(item) for item in value] + if value is None or isinstance(value, (bool, int, float)): + return value + return memory_core.redact(value) + + +def _spool_path() -> Path: + return memory_core.data_dir() / "telemetry.jsonl" + + +def _identity_path() -> Path: + return memory_core.data_dir() / "telemetry-identity.json" + + +def _read_identity() -> dict[str, str]: + try: + value = json.loads(_identity_path().read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + return value if isinstance(value, dict) else {} + + +def _write_identity(identity: dict[str, str]) -> None: + path = _identity_path() + temporary = path.with_suffix(f".{os.getpid()}.tmp") + try: + path.parent.mkdir(parents=True, exist_ok=True) + temporary.write_text(json.dumps(identity), encoding="utf-8") + temporary.replace(path) + except OSError: + try: + temporary.unlink() + except OSError: + pass + + +def anonymous_id(identity: dict[str, str] | None = None) -> str: + """Per-machine anonymous identifier, created and persisted on first use.""" + identity = _read_identity() if identity is None else identity + existing = identity.get("anonymous_id") + if existing: + return existing + created = f"code-anon-{uuid.uuid4().hex}" + identity["anonymous_id"] = created + _write_identity(identity) + return created + + +def is_first_run() -> bool: + """Whether this machine has never recorded a plugin event before.""" + return not _identity_path().exists() + + +def record( + event: str, + *, + repo: Any = None, + session_id: str | None = None, + **properties: Any, +) -> None: + """Append one event to the local spool. Never blocks and never raises.""" + if not is_enabled(): + return + try: + spool = _spool_path() + try: + if spool.stat().st_size > SPOOL_LIMIT_BYTES: + return + except OSError: + pass + properties = _safe_value(properties) + properties.update( + harness=_harness, + plugin_version=memory_core.PLUGIN_VERSION, + os=sys.platform, + python_version=platform.python_version(), + ) + if repo is not None: + properties["repo_hash"] = _digest(getattr(repo, "identity", "")) + if session_id: + properties["session_hash"] = _digest(session_id) + line = json.dumps( + { + "event": f"{EVENT_PREFIX}.{event}", + "timestamp": memory_core.utc_now(), + "properties": { + key: value for key, value in properties.items() if value is not None + }, + }, + separators=(",", ":"), + default=str, + ) + spool.parent.mkdir(parents=True, exist_ok=True) + with spool.open("a", encoding="utf-8") as handle: + handle.write(line + "\n") + except Exception: + pass + + +def error_kind(exc: BaseException | str) -> str: + """Coarse, content-free label for a failure, safe to send.""" + text = exc if isinstance(exc, str) else f"{type(exc).__name__}: {exc}" + lowered = text.lower() + if "timed out" in lowered or "timeout" in lowered: + return "timeout" + if "401" in lowered or "403" in lowered or "unauthor" in lowered or "forbidden" in lowered: + return "auth" + if "429" in lowered or "rate limit" in lowered: + return "rate-limited" + if any(code in lowered for code in ("500", "502", "503", "504")): + return "server-error" + if "400" in lowered or "422" in lowered: + return "bad-request" + if isinstance(exc, str): + return "other" + if isinstance(exc, urllib.error.URLError): + return "network" + return type(exc).__name__ + + +def spawn_flush() -> bool: + """Start the detached sender that drains the spool.""" + if not is_enabled(): + return False + try: + if not _spool_path().exists() and not any( + memory_core.data_dir().glob("telemetry-*.sending") + ): + return False + subprocess.Popen( + [sys.executable, str(Path(__file__).resolve())], + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + close_fds=True, + **memory_core.detached_process_kwargs(), + ) + return True + except Exception: + return False + + +def _claim_spool() -> Path | None: + """Rename the spool aside so exactly one sender owns each batch.""" + directory = memory_core.data_dir() + claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending" + spool = _spool_path() + try: + spool.replace(claim) + return claim + except OSError: + pass + now = time.time() + for orphan in sorted(directory.glob("telemetry-*.sending")): + try: + age = now - orphan.stat().st_mtime + except OSError: + continue + if age > CLAIM_EXPIRY_SECONDS: + try: + orphan.unlink() + except OSError: + pass + continue + if age < CLAIM_STALE_SECONDS: + continue + try: + orphan.replace(claim) + return claim + except OSError: + continue + return None + + +def _resolve_email(key: str) -> str: + """Trade the API key for the account email so events join other Mem0 surfaces.""" + url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/" + request = urllib.request.Request( + url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json"} + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT) as response: + payload = json.loads(response.read().decode("utf-8")) + except Exception: + return "" + email = payload.get("user_email") if isinstance(payload, dict) else "" + return email if isinstance(email, str) else "" + + +def _post(payload: dict[str, Any], url: str) -> bool: + request = urllib.request.Request( + url, + data=json.dumps(payload, default=str).encode("utf-8"), + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=SEND_TIMEOUT): + return True + except Exception: + return False + + +def resolve_distinct_id() -> tuple[str, str]: + """Return the PostHog distinct id and the anonymous id it replaced, if any.""" + identity = _read_identity() + email = identity.get("email", "") + if email: + return email, "" + key = memory_core.api_key() + if not key: + return anonymous_id(identity), "" + email = _resolve_email(key) + if not email: + return anonymous_id(identity), "" + previous = identity.get("anonymous_id", "") + identity["email"] = email + _write_identity(identity) + return email, previous + + +def flush() -> int: + """Drain claimed spools to PostHog and return the number of events sent.""" + if not is_enabled(): + return 0 + claim = _claim_spool() + if claim is None: + return 0 + try: + lines = claim.read_text(encoding="utf-8").splitlines() + except OSError: + return 0 + events = [] + for line in lines: + try: + value = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(value, dict) and value.get("event"): + events.append(value) + if not events: + try: + claim.unlink() + except OSError: + pass + return 0 + + distinct_id, aliased_anonymous_id = resolve_distinct_id() + if aliased_anonymous_id: + _post( + { + "api_key": POSTHOG_API_KEY, + "event": "$identify", + "distinct_id": distinct_id, + "properties": { + "$anon_distinct_id": aliased_anonymous_id, + "$lib": "posthog-python", + }, + }, + POSTHOG_CAPTURE_URL, + ) + + sent = 0 + for start in range(0, len(events), BATCH_SIZE): + batch = [ + { + "event": event["event"], + "distinct_id": distinct_id, + "timestamp": event.get("timestamp"), + "properties": { + "source": _source_tag, + "language": "python", + "$process_person_profile": False, + "$lib": "posthog-python", + **(event.get("properties") or {}), + }, + } + for event in events[start : start + BATCH_SIZE] + ] + if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL): + return sent + sent += len(batch) + try: + claim.unlink() + except OSError: + pass + return sent + + +def main() -> int: + flush() + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception: + raise SystemExit(0) diff --git a/integrations/mem0-agent-plugin/mcp.json b/integrations/mem0-agent-plugin/mcp.json new file mode 100644 index 000000000..dffd6771f --- /dev/null +++ b/integrations/mem0-agent-plugin/mcp.json @@ -0,0 +1,12 @@ +{ + "$schema": "https://agent-plugins.org/schemas/1.0.0/mcp.schema.json", + "mcpServers": { + "mem0": { + "type": "stdio", + "command": "python3", + "args": [ + "${PLUGIN_ROOT}/core/mcp_server.py" + ] + } + } +} diff --git a/integrations/mem0-agent-plugin/plugin.json b/integrations/mem0-agent-plugin/plugin.json new file mode 100644 index 000000000..2cbaa99de --- /dev/null +++ b/integrations/mem0-agent-plugin/plugin.json @@ -0,0 +1,19 @@ +{ + "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", + "name": "mem0", + "version": "0.3.1", + "description": "Cross-session memory and token savings for coding agents.", + "author": { + "name": "Mem0", + "email": "support@mem0.ai" + }, + "homepage": "https://docs.mem0.ai/integrations", + "repository": "https://github.com/mem0ai/mem0", + "license": "Apache-2.0", + "keywords": [ + "memory", + "coding-agents", + "continual-learning", + "token-efficiency" + ] +} diff --git a/integrations/mem0-agent-plugin/skills/forget/SKILL.md b/integrations/mem0-agent-plugin/skills/forget/SKILL.md new file mode 100644 index 000000000..0c272e9ff --- /dev/null +++ b/integrations/mem0-agent-plugin/skills/forget/SKILL.md @@ -0,0 +1,25 @@ +--- +name: forget +description: Delete the Mem0 memories stored for this repository and this user. Use when the user asks to forget, clear, wipe, or delete memories. +--- + +# Forget this repository's memories + +This permanently deletes remote memories. Before running anything, tell the +user exactly what will be deleted: their own memories for this repository +only. The repository's project memory is shared by everyone who works in it, +so it stays unless the user explicitly asks to delete that too. + +After the user confirms, run: + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "coding-agent" forget --remote --yes +``` + +If the user also asked to delete the repository's shared project memory, add +`--include-project-memory` and say that this removes it for every teammate. + +Report what the command output says was deleted. If the user only wants local +data cleared (evidence log, pending queue), run the same command without +`--remote`. Never pass `--yes` before the user has confirmed in this +conversation. diff --git a/integrations/mem0-agent-plugin/skills/pause/SKILL.md b/integrations/mem0-agent-plugin/skills/pause/SKILL.md new file mode 100644 index 000000000..c6316dd0b --- /dev/null +++ b/integrations/mem0-agent-plugin/skills/pause/SKILL.md @@ -0,0 +1,19 @@ +--- +name: pause +description: Pause Mem0 memory capture on this machine. Use when the user wants to stop memories being recorded, for example for private work or experiments. +--- + +# Pause memory capture + +To pause (hooks stop capturing and sending session content; a minimal +anonymous telemetry ping still fires at session start unless +`MEM0_TELEMETRY=false`): + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "coding-agent" pause +``` + +Confirm the new state back to the user, and remind them that already-created +memories still exist and remain searchable. Pending unsent packets are held +while paused, not expired, and are delivered after resuming. To turn capture +back on, use `/mem0:resume`. diff --git a/integrations/mem0-agent-plugin/skills/remember/SKILL.md b/integrations/mem0-agent-plugin/skills/remember/SKILL.md new file mode 100644 index 000000000..531a8b70b --- /dev/null +++ b/integrations/mem0-agent-plugin/skills/remember/SKILL.md @@ -0,0 +1,20 @@ +--- +name: remember +description: Acknowledge a "remember this" request and make sure it is captured well. Use when the user explicitly asks to remember, note, or save something for future sessions. +--- + +# Remember something for future sessions + +Mem0 creates memories from the session automatically — there is no separate +write command. When the user asks to remember something: + +1. Restate the fact clearly and completely in your reply, in one or two + sentences, including any names, values, or paths it depends on. Your visible + reply is what memory extraction reads, so a precise restatement is what gets + remembered. +2. Tell the user it will be saved with this session's memories when the session + ends or compacts, and that it will surface in future sessions in this + repository (they can check later with /mem0:search). + +Do not invent a storage confirmation or a memory ID — creation happens in the +background after the session. diff --git a/integrations/mem0-agent-plugin/skills/resume/SKILL.md b/integrations/mem0-agent-plugin/skills/resume/SKILL.md new file mode 100644 index 000000000..3668d68dd --- /dev/null +++ b/integrations/mem0-agent-plugin/skills/resume/SKILL.md @@ -0,0 +1,18 @@ +--- +name: resume +description: Resume Mem0 memory capture after it was paused with /mem0:pause. +--- + +# Resume memory capture + +Resume memory capture for this machine. + +Run: + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "coding-agent" resume +``` + +Confirm to the user that capture is active again. New sessions record evidence and +create memories as normal; nothing that happened while paused is retroactively +captured. diff --git a/integrations/mem0-agent-plugin/skills/search/SKILL.md b/integrations/mem0-agent-plugin/skills/search/SKILL.md new file mode 100644 index 000000000..d52c4ff34 --- /dev/null +++ b/integrations/mem0-agent-plugin/skills/search/SKILL.md @@ -0,0 +1,26 @@ +--- +name: search +description: Search memories from earlier Coding Agent sessions in this repository. Use it when earlier work may already explain the code, error, decision, or command you need, so you can avoid repeating file reads, searches, or experiments. +--- + +# Search memories + +Call `search_memories` with the user's question. Treat `--top-k`, `--category`, +`--scope`, and `--run-id` as tool arguments instead of including them in the +query. + +Omit `top_k` to use Mem0's configured default. Omit `category` to search every +category; a category is a best-effort label Mem0 assigned when it saved the +memory, so if a category search misses, repeat it without the category. Omit +`scope` to use the configured default, normally `repo`: this repository's +shared memory, which everyone who works in it contributes to, plus your own +preferences. + +Pass `scope` when the question needs something else: `dir` to narrow the +shared memory to the directory you are working in (a package inside a +monorepo), `mine` for your own preferences alone. + +Pass `run_id` with any scope to retrieve memories saved in a specific coding-agent +session. Omit `run_id` to search across sessions. It filters the memories returned; +it does not identify the session making the search request. Use a known session ID, +never invent one. Return the tool's result directly. diff --git a/integrations/mem0-agent-plugin/skills/status/SKILL.md b/integrations/mem0-agent-plugin/skills/status/SKILL.md new file mode 100644 index 000000000..7717e656b --- /dev/null +++ b/integrations/mem0-agent-plugin/skills/status/SKILL.md @@ -0,0 +1,22 @@ +--- +name: status +description: Show whether Mem0 memory is working in this repository, covering configuration, capture state, pending flushes, and whether the Mem0 API key is valid. Use when the user asks whether memory is on, why a memory is missing, or anything looks broken. +--- + +# Memory status + +Run both commands and report the combined result in plain language: + +```bash +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "coding-agent" status --json +python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "coding-agent" doctor +``` + +Summarize, using only fields the JSON actually reports: whether capture is +active or paused, the user ID and repository scope (`repo_id`), whether an +API key is configured, the event/flush/retrieval counts (`flushes` is the +number of completed flushes, not a pending count), and the doctor check +results. If doctor reports an authentication failure (401 / invalid key), say +clearly that the Mem0 API key is invalid or expired and that memories are NOT +being created. Never report an auth failure as "no memories found". Suggest +reinstalling with `--config api_key=...` in that case. diff --git a/integrations/mem0-plugin/.codex-mcp.json b/integrations/mem0-plugin/.codex-mcp.json deleted file mode 100644 index eba6c6609..000000000 --- a/integrations/mem0-plugin/.codex-mcp.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "mcpServers": { - "mem0": { - "url": "https://mcp.mem0.ai/mcp", - "bearer_token_env_var": "MEM0_API_KEY" - } - } -} diff --git a/integrations/mem0-plugin/.codex-plugin/plugin.json b/integrations/mem0-plugin/.codex-plugin/plugin.json deleted file mode 100644 index cfdd9a6c0..000000000 --- a/integrations/mem0-plugin/.codex-plugin/plugin.json +++ /dev/null @@ -1,35 +0,0 @@ -{ - "name": "mem0", - "version": "0.2.15", - "description": "Persistent memory for Codex. Remembers decisions, patterns, and preferences across sessions.", - "author": { - "name": "Mem0", - "email": "support@mem0.ai", - "url": "https://mem0.ai" - }, - "homepage": "https://mem0.ai", - "repository": "https://github.com/mem0ai/mem0", - "license": "Apache-2.0", - "keywords": ["memory", "personalization", "mcp", "semantic-search"], - "skills": "./skills/", - "mcpServers": "./.codex-mcp.json", - "hooks": "./hooks/codex-hooks.json", - "interface": { - "displayName": "Mem0", - "shortDescription": "Persistent memory layer for AI coding workflows", - "longDescription": "Mem0 adds long-term memory to Codex. Store decisions, user preferences, project context, and session state across conversations. Memories are automatically retrieved via semantic search so Codex always has the right context.", - "developerName": "Mem0", - "category": "Productivity", - "capabilities": ["Read", "Write"], - "websiteURL": "https://mem0.ai", - "privacyPolicyURL": "https://mem0.ai/privacy", - "termsOfServiceURL": "https://mem0.ai/terms", - "defaultPrompt": [ - "Search my memories for recent project decisions", - "Remember that I prefer TypeScript over JavaScript", - "What do you know about my coding preferences?" - ], - "brandColor": "#FBBF24", - "logo": "./logo.svg" - } -} diff --git a/integrations/mem0-plugin/.cursor-mcp.json b/integrations/mem0-plugin/.cursor-mcp.json deleted file mode 100644 index 80c40be60..000000000 --- a/integrations/mem0-plugin/.cursor-mcp.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "mcpServers": { - "mem0": { - "url": "https://mcp.mem0.ai/mcp/", - "headers": { - "Authorization": "Token ${env:MEM0_API_KEY}" - } - } - } -} diff --git a/integrations/mem0-plugin/.cursor-plugin/plugin.json b/integrations/mem0-plugin/.cursor-plugin/plugin.json deleted file mode 100644 index fe500f4cd..000000000 --- a/integrations/mem0-plugin/.cursor-plugin/plugin.json +++ /dev/null @@ -1,17 +0,0 @@ -{ - "name": "mem0", - "version": "0.2.15", - "description": "Mem0 memory layer for AI applications. Add persistent memory, personalization, and semantic search using the Mem0 Platform MCP server.", - "author": { - "name": "Mem0", - "email": "support@mem0.ai" - }, - "homepage": "https://mem0.ai", - "repository": "https://github.com/mem0ai/mem0", - "logo": "logo.svg", - "license": "Apache-2.0", - "keywords": ["mem0", "memory", "mcp", "personalization", "semantic-search"], - "skills": "./skills/", - "hooks": "./hooks/cursor-hooks.json", - "mcpServers": "./.cursor-mcp.json" -} diff --git a/integrations/mem0-plugin/.kimi-plugin/plugin.json b/integrations/mem0-plugin/.kimi-plugin/plugin.json deleted file mode 100644 index 27d8a8ae4..000000000 --- a/integrations/mem0-plugin/.kimi-plugin/plugin.json +++ /dev/null @@ -1,86 +0,0 @@ -{ - "name": "mem0", - "version": "0.1.0", - "description": "Persistent memory for Kimi Code. Remembers decisions, patterns, and preferences across sessions.", - "author": { - "name": "Mem0", - "email": "support@mem0.ai", - "url": "https://mem0.ai" - }, - "homepage": "https://mem0.ai", - "repository": "https://github.com/mem0ai/mem0", - "license": "Apache-2.0", - "keywords": ["memory", "personalization", "mcp", "semantic-search"], - "skills": "./skills/", - "sessionStart": { - "skill": "context-loader" - }, - "hooks": [ - { - "event": "SessionStart", - "matcher": "^(startup|resume)$", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" on_session_start.sh", - "timeout": 30 - }, - { - "event": "UserPromptSubmit", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" on_user_prompt.sh", - "timeout": 12 - }, - { - "event": "PreToolUse", - "matcher": "^(Write|Edit|MultiEdit)$", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" block_memory_write.sh", - "timeout": 5 - }, - { - "event": "PreToolUse", - "matcher": "^mcp__.*mem0.*__(add_memory|search_memories|get_memories|delete_all_memories)$", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" enforce_metadata_defaults.sh", - "timeout": 5 - }, - { - "event": "PreToolUse", - "matcher": "^Read$", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" on_file_read.sh", - "timeout": 5 - }, - { - "event": "PostToolUse", - "matcher": "^mcp__.*mem0.*__", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" on_post_tool_use.sh", - "timeout": 5 - }, - { - "event": "PostToolUse", - "matcher": "^Bash$", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" on_bash_output.sh", - "timeout": 12 - }, - { - "event": "Stop", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" on_stop.sh", - "timeout": 30 - }, - { - "event": "PreCompact", - "command": "\"$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh\" on_pre_compact.sh", - "timeout": 30 - } - ], - "mcpServers": { - "mem0": { - "transport": "http", - "url": "https://mcp.mem0.ai/mcp/", - "bearerTokenEnvVar": "MEM0_API_KEY" - } - }, - "interface": { - "displayName": "Mem0", - "shortDescription": "Persistent memory layer for AI coding workflows", - "longDescription": "Mem0 adds long-term memory to Kimi Code. Store decisions, user preferences, project context, and session state across conversations. Memories are automatically retrieved via semantic search so Kimi always has the right context available.", - "developerName": "Mem0", - "websiteURL": "https://mem0.ai", - "logo": "./logo.svg" - } -} diff --git a/integrations/mem0-plugin/.opencode-plugin/dream.test.ts b/integrations/mem0-plugin/.opencode-plugin/dream.test.ts deleted file mode 100644 index f8fe588dc..000000000 --- a/integrations/mem0-plugin/.opencode-plugin/dream.test.ts +++ /dev/null @@ -1,100 +0,0 @@ -import { afterEach, beforeEach, describe, expect, test } from "bun:test"; -import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; -import { tmpdir } from "node:os"; -import { join } from "node:path"; -import { - loadDreamConfig, - incrementSessionCount, - checkCheapGates, - checkMemoryGate, - acquireDreamLock, - releaseDreamLock, - recordDreamCompletion, - DREAM_DEFAULTS, - DREAM_PROTOCOL, -} from "./dream"; - -let dir: string; - -beforeEach(() => { - dir = mkdtempSync(join(tmpdir(), "mem0-dream-")); -}); - -afterEach(() => { - try { - rmSync(dir, { recursive: true, force: true }); - } catch { - /* ignore */ - } - delete process.env.MEM0_DREAM; -}); - -describe("auto-dream gates", () => { - test("memory gate passes at >= minMemories, fails below", () => { - expect(checkMemoryGate(DREAM_DEFAULTS.minMemories, {}).pass).toBe(true); - expect(checkMemoryGate(DREAM_DEFAULTS.minMemories - 1, {}).pass).toBe(false); - }); - - test("cheap gates: fresh state blocks on session count, passes after enough sessions", () => { - // Fresh state: time gate passes (lastConsolidatedAt=0), but 0 sessions blocks. - expect(checkCheapGates(dir, {}).proceed).toBe(false); - for (let i = 0; i < DREAM_DEFAULTS.minSessions; i++) { - incrementSessionCount(dir, `ses_${i}`); - } - expect(checkCheapGates(dir, {}).proceed).toBe(true); - }); - - test("incrementSessionCount only counts distinct session ids", () => { - incrementSessionCount(dir, "ses_a"); - incrementSessionCount(dir, "ses_a"); - incrementSessionCount(dir, "ses_a"); - expect(checkCheapGates(dir, { minHours: 0 }).reason).toContain("sessions: 1"); - }); - - test("recordDreamCompletion resets gates (recent time blocks again)", () => { - for (let i = 0; i < 6; i++) incrementSessionCount(dir, `ses_${i}`); - expect(checkCheapGates(dir, {}).proceed).toBe(true); - recordDreamCompletion(dir); - const r = checkCheapGates(dir, {}); - expect(r.proceed).toBe(false); - expect(r.reason).toContain("time"); - }); - - test("dream lock is exclusive and reclaimable after release", () => { - expect(acquireDreamLock(dir)).toBe(true); - expect(acquireDreamLock(dir)).toBe(false); - releaseDreamLock(dir); - expect(acquireDreamLock(dir)).toBe(true); - }); -}); - -describe("dream config", () => { - test("defaults when no settings file", () => { - const cfg = loadDreamConfig(dir); - expect(cfg.enabled).toBe(true); - expect(cfg.auto).toBe(true); - expect(cfg.minMemories).toBe(DREAM_DEFAULTS.minMemories); - }); - - test("MEM0_DREAM=false force-disables", () => { - process.env.MEM0_DREAM = "false"; - expect(loadDreamConfig(dir).enabled).toBe(false); - }); - - test("settings.json dream block overrides defaults", () => { - writeFileSync( - join(dir, "settings.json"), - JSON.stringify({ dream: { minMemories: 99, auto: false } }), - ); - const cfg = loadDreamConfig(dir); - expect(cfg.minMemories).toBe(99); - expect(cfg.auto).toBe(false); - expect(cfg.enabled).toBe(true); - }); - - test("protocol uses native tools, not the MCP tool", () => { - expect(DREAM_PROTOCOL).toContain("get_memories"); - expect(DREAM_PROTOCOL).toContain("add_memory"); - expect(DREAM_PROTOCOL).not.toContain("mem0_memory"); - }); -}); diff --git a/integrations/mem0-plugin/.opencode-plugin/dream.ts b/integrations/mem0-plugin/.opencode-plugin/dream.ts deleted file mode 100644 index 68395c54b..000000000 --- a/integrations/mem0-plugin/.opencode-plugin/dream.ts +++ /dev/null @@ -1,225 +0,0 @@ -/** - * Auto-dream: gated automatic memory consolidation for the Mem0 OpenCode plugin. - * - * Ported from the (stable) pi-agent plugin's dream module and adapted to - * OpenCode's hook model. When the cheap gates (time since last consolidation + - * sessions since) and the memory-count gate all pass, the plugin injects the - * DREAM_PROTOCOL into the agent's context so it consolidates memories (merge - * duplicates, drop stale/sensitive entries, rewrite vague ones) before - * answering. A filesystem lock prevents concurrent sessions from dreaming at - * once, and completion is recorded so it won't re-trigger until the next cycle. - * - * State + lock live in ~/.mem0/ alongside settings.json. Opt out with - * MEM0_DREAM=false, or tune via the `dream` block in ~/.mem0/settings.json. - */ - -import { existsSync, mkdirSync, readFileSync, writeFileSync, unlinkSync } from "node:fs"; -import { join } from "node:path"; - -export interface DreamConfig { - enabled: boolean; - auto: boolean; - minHours: number; - minSessions: number; - minMemories: number; -} - -interface DreamState { - lastConsolidatedAt: number; - sessionsSince: number; - lastSessionId: string | null; -} - -interface DreamLock { - pid: number; - startedAt: number; -} - -const LOCK_STALE_MS = 60 * 60 * 1000; - -export const DREAM_DEFAULTS: DreamConfig = { - enabled: true, - auto: true, - minHours: 24, - minSessions: 5, - minMemories: 20, -}; - -function statePath(stateDir: string): string { - return join(stateDir, "mem0-dream-state.json"); -} - -function lockPath(stateDir: string): string { - return join(stateDir, "mem0-dream.lock"); -} - -function ensureDir(dir: string): void { - try { - mkdirSync(dir, { recursive: true }); - } catch { - /* exists */ - } -} - -function readState(stateDir: string): DreamState { - try { - return JSON.parse(readFileSync(statePath(stateDir), "utf-8")) as DreamState; - } catch { - return { lastConsolidatedAt: 0, sessionsSince: 0, lastSessionId: null }; - } -} - -function writeState(stateDir: string, state: DreamState): void { - ensureDir(stateDir); - writeFileSync(statePath(stateDir), JSON.stringify(state, null, 2)); -} - -/** - * Load dream config from ~/.mem0/settings.json (`dream` block), applying - * defaults. MEM0_DREAM=false (or 0/no/off) force-disables regardless. - */ -export function loadDreamConfig(settingsDir: string): DreamConfig { - let envEnabled: boolean | undefined; - const env = process.env.MEM0_DREAM; - if (env !== undefined) { - const s = env.toLowerCase(); - envEnabled = s !== "false" && s !== "0" && s !== "no" && s !== "off"; - } - - let cfg: DreamConfig = { ...DREAM_DEFAULTS }; - try { - const sp = join(settingsDir, "settings.json"); - if (existsSync(sp)) { - const settings = JSON.parse(readFileSync(sp, "utf-8")); - const d = settings?.dream; - if (d && typeof d === "object") { - cfg = { - enabled: typeof d.enabled === "boolean" ? d.enabled : cfg.enabled, - auto: typeof d.auto === "boolean" ? d.auto : cfg.auto, - minHours: typeof d.minHours === "number" ? d.minHours : cfg.minHours, - minSessions: typeof d.minSessions === "number" ? d.minSessions : cfg.minSessions, - minMemories: typeof d.minMemories === "number" ? d.minMemories : cfg.minMemories, - }; - } - } - } catch { - /* defaults */ - } - - if (envEnabled !== undefined) cfg.enabled = envEnabled; - return cfg; -} - -/** Count a new session toward the dream gate (once per distinct sessionId). */ -export function incrementSessionCount(stateDir: string, sessionId: string): void { - const state = readState(stateDir); - if (state.lastSessionId !== sessionId) { - state.sessionsSince++; - state.lastSessionId = sessionId; - writeState(stateDir, state); - } -} - -/** Cheap gates that don't need an API call: time since last + sessions since. */ -export function checkCheapGates( - stateDir: string, - config: Partial, -): { proceed: boolean; reason?: string } { - const minHours = config.minHours ?? DREAM_DEFAULTS.minHours; - const minSessions = config.minSessions ?? DREAM_DEFAULTS.minSessions; - const state = readState(stateDir); - - const hoursSince = (Date.now() - state.lastConsolidatedAt) / 3_600_000; - if (hoursSince < minHours) { - return { proceed: false, reason: `time: ${hoursSince.toFixed(1)}h < ${minHours}h` }; - } - if (state.sessionsSince < minSessions) { - return { proceed: false, reason: `sessions: ${state.sessionsSince} < ${minSessions}` }; - } - return { proceed: true }; -} - -/** Memory-count gate (uses the count already fetched at session init). */ -export function checkMemoryGate( - memoryCount: number, - config: Partial, -): { pass: boolean; reason?: string } { - const minMemories = config.minMemories ?? DREAM_DEFAULTS.minMemories; - if (memoryCount < minMemories) { - return { pass: false, reason: `memories: ${memoryCount} < ${minMemories}` }; - } - return { pass: true }; -} - -/** Acquire an exclusive dream lock (stale locks > 1h are reclaimed). */ -export function acquireDreamLock(stateDir: string): boolean { - ensureDir(stateDir); - const lp = lockPath(stateDir); - - try { - const lock = JSON.parse(readFileSync(lp, "utf-8")) as DreamLock; - if (Date.now() - lock.startedAt < LOCK_STALE_MS) { - return false; - } - try { - unlinkSync(lp); - } catch { - /* race ok */ - } - } catch { - /* no lock file */ - } - - const lock: DreamLock = { pid: process.pid, startedAt: Date.now() }; - try { - writeFileSync(lp, JSON.stringify(lock), { flag: "wx" }); - return true; - } catch { - return false; - } -} - -export function releaseDreamLock(stateDir: string): void { - try { - unlinkSync(lockPath(stateDir)); - } catch { - /* already gone */ - } -} - -/** Reset the gates after a successful consolidation. */ -export function recordDreamCompletion(stateDir: string): void { - const state = readState(stateDir); - state.lastConsolidatedAt = Date.now(); - state.sessionsSince = 0; - state.lastSessionId = null; - writeState(stateDir, state); -} - -/** - * Consolidation protocol injected into the agent context when a dream is - * triggered. Uses the plugin's native OpenCode memory tools (get_memories / - * add_memory / delete_memory) rather than an MCP tool. - */ -export const DREAM_PROTOCOL = ` -You are running memory consolidation. Complete these steps using the mem0 memory tools (get_memories, add_memory, delete_memory): - -1. ORIENT — Call get_memories to list all memories. Count by category. Note oldest/newest. - -2. GATHER TARGETS — Review each memory. Classify as: - - DELETE: sensitive information (API keys, passwords, tokens), expired/stale entries, noise, redundant operational details - - MERGE: near-duplicates (same fact stated differently). Keep the better-worded one, delete the other. - - REWRITE: vague, first-person, or poorly-categorized entries. add_memory with improved text, then delete_memory the old one. - - KEEP: everything else. - Skip any memory starting with "[PINNED]". - -3. CONSOLIDATE — Execute the changes: - - Delete stale/duplicate entries with delete_memory - - For merges: add_memory the merged text, delete_memory both originals - - For rewrites: add_memory the improved version, delete_memory the original - -4. REPORT — Summarize: how many reviewed, deleted, merged, rewritten, final count. - -Quality targets: zero sensitive data stored, zero duplicates, all entries are atomic (one fact each), 15-50 words each. -After consolidation, respond to the user's message normally. -`; diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-dream/SKILL.md b/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-dream/SKILL.md deleted file mode 100644 index 47441e0f4..000000000 --- a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-dream/SKILL.md +++ /dev/null @@ -1,237 +0,0 @@ ---- -name: mem0-dream -description: Consolidates stored memories by merging duplicates, resolving contradictions, and pruning stale entries. Use when memory count is high, search results feel noisy or repetitive, or periodic cleanup is needed to maintain memory quality. ---- - -# Mem0 Dream — Memory Consolidation - -This skill performs a memory consolidation pass: it fetches all project memories, -identifies near-duplicates, flags contradictions, and prunes stale entries based on -configured retention policies. All proposed changes are shown as a diff for user -approval before anything is modified. - -**IMPORTANT: Execute steps strictly in order (1 → 2 → 3 → 4 → 5 → 6). Each step depends on the previous one. Do NOT run steps in parallel or skip ahead.** - -## Step 1: Load Retention Policies - -Check for a project config file in the project root (current working directory): - -1. Look for `.mem0.json` first. If it exists, parse it as JSON and read the - `retention` field (a dict of `category → days | null`). -2. If `.mem0.json` is not present, look for `.mem0.md`. If it exists, scan it - for a `retention:` section or YAML front matter with retention settings and - parse what you find. -3. If neither file exists, skip config loading entirely. - -If no config is found or the config contains no retention settings, fall back to -these built-in defaults: - -| `metadata.type` | Default retention | -|---|---| -| `session_state` | 90 days | -| `compact_summary` | 90 days | -| all others | no pruning | - -Store the resolved policies for use in Step 3. - ---- - -## Step 2: Fetch ALL Project Memories - -Call `get_memories` to retrieve every memory for the active project: - -```python -get_memories( - filters={"AND": [{"user_id": ""}, {"app_id": ""}]}, - page_size=200, -) -``` - -If the response indicates more pages exist, paginate until all memories are fetched. -Collect the full list before proceeding. If zero memories are found, print: - -``` -No memories found for project . Nothing to consolidate. -``` - -…and stop. - ---- - -## Step 3: Analyze — Find Issues - -Work entirely in-memory; do not modify anything yet. - -Group memories by `metadata.type` (use `"unknown"` when the field is absent). -For each group, identify the following: - -### 3a. Near-duplicate pairs (merge candidates) - -Two memories are near-duplicates when they express the same fact or decision but -phrased differently (e.g., "Use PostgreSQL for auth" and "Auth DB is PostgreSQL"). - -Heuristics — two memories are near-duplicates if **all** of these hold: -- Similarity threshold: estimated cosine similarity > 0.9 (use noun/keyword overlap as proxy — if >60% of significant nouns overlap, treat as >0.9 similarity). -- Same `metadata.type`. -- Neither memory is pinned (`metadata.pinned != true`). - -For each qualifying pair, draft a merged version that is more complete and specific -than either original. - -### 3b. Contradictions - -Two memories contradict when they assert opposing facts about the same topic -(e.g., "Deploy to ECS" vs. "Deploy to Vercel"). - -Identify the likely winner: the more recent memory with higher confidence wins. -Store both IDs and their content for user review. - -### 3c. Prune candidates - -A memory is a prune candidate when **any** of the following is true: - -1. Its `metadata.type` has a retention policy and the memory is older than the - configured number of days (compare `created_at` to today). -2. Its confidence score is below 0.3 AND it contains no information unique to - this project (no file paths, identifiers, or domain-specific nouns). - -**Always skip memories where `metadata.pinned == true`**, regardless of age or -confidence. - ---- - -## Step 4: Print Diff Report - -Print a structured diff to the terminal before making any changes. Use exactly -this format: - -``` -## dream — consolidation report - -Merges (): - [mem0:] + [mem0:] → "" - -Conflicts (): - [mem0:] vs [mem0:] — "" [A/B/skip] - -Prune (): - [mem0:] — , d old - -Proposed: merges, prunes, conflicts. Apply? [Y/n] -``` - -If there are zero items in any category, omit that section entirely. - -If there are zero total proposals (no merges, no prunes, no conflicts), print: - -``` -Dream complete. No duplicate, contradictory, or stale memories found. -``` - -…and stop. - ---- - -## Step 5: Wait for User Input and Apply - -### 5a. Contradictions - -For each `CONFLICT` pair in the report, wait for the user to type `A`, `B`, or -`skip` (case-insensitive). If they enter nothing (empty), treat as `skip`. - -Record the winner for each pair before proceeding to the final apply confirmation. - -### 5b. Final confirmation - -After all conflict resolutions are collected, prompt: - -``` -Apply? [Y/n] -``` - -If the user types `n` or `no` (case-insensitive), print `Cancelled. No changes made.` -and stop. - -If the user confirms (`Y`, `yes`, or empty / Enter), apply all changes in this order: - -#### Merges - -For each approved merge pair: -1. `delete_memory()` -2. `delete_memory()` -3. `add_memory` with: - - `text=""` - - `user_id=` - - `app_id=` (top-level, not in metadata) - - `metadata={"type": "", "branch": "", "confidence": , "source": "mem0-dream"}` - - `infer=False` - -#### Contradictions (resolved) - -For each resolved conflict where the user chose A or B: -- Delete the loser (the non-chosen memory): `delete_memory(memory_id=)` - -Contradictions where the user chose `skip` are left untouched. - -#### Prunes - -For each prune candidate: -- `delete_memory()` - ---- - -## Step 6: Print Summary - -After all changes are applied, print: - -``` -Dream complete — merged: , pruned: , conflicts resolved: , skipped: -``` - ---- - -## Auto mode - -When invoked with `--auto` (e.g., `/mem0-dream --auto`), run non-interactively: - -- **Merges**: applied automatically (no contradiction, both are compatible). -- **Prunes**: applied automatically (age/confidence-based, no ambiguity). -- **Contradictions**: skipped — they require human judgment. - -### Concurrency guard - -Before doing any work, check for a lock file at `/tmp/mem0_dream_auto.lock`: -- If the lock file exists and is less than 10 minutes old, print `[mem0-dream --auto] Another run in progress — skipping.` and stop. -- Otherwise, create the lock file (write the current timestamp). Delete it when done (in all exit paths). - -### Execution - -In auto mode: -1. Load policies and fetch memories (Steps 1–3) as normal. -2. Apply merges and prunes silently without printing the diff or prompting. -3. Print a compact summary: - ``` - [mem0-dream --auto] project= merged= pruned= conflicts_skipped= - ``` -4. If contradictions were detected but skipped, check if a `mem0-dream-auto` reminder already exists before storing one: - - Search for existing reminders: `search_memories(query="mem0-dream contradictions manual review", filters={"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"source": "mem0-dream-auto"}}]}, top_k=1)` - - If a result exists with similarity > 0.9, skip storing the reminder (one already exists). - - If no match, store the reminder: - ```python - add_memory( - text="mem0-dream detected contradiction(s) requiring manual review. Run /mem0-dream to resolve them interactively.", - user_id="", - app_id="", - metadata={"type": "task_learning", "source": "mem0-dream-auto", "branch": ""}, - infer=False, - ) - ``` - -## See also - -- `/mem0-forget` — targeted deletion of specific memories (search + confirm + delete) -- `/mem0-status --deep` — quick quality scan without applying changes - -## Output formatting - -IMPORTANT: Do NOT use markdown in your output. OpenCode TUI renders text verbatim — markdown like **bold**, ## headers, and | table | syntax appears as raw characters. Use plain text with indentation for structure. Use dashes for lists. Use spaces to align columns instead of markdown tables. diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-pin/SKILL.md b/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-pin/SKILL.md deleted file mode 100644 index e2b86ebc5..000000000 --- a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-pin/SKILL.md +++ /dev/null @@ -1,71 +0,0 @@ ---- -name: mem0-pin -description: Pins or unpins a memory to protect it from pruning during dream consolidation. Use when a memory is critical and must never be removed, such as architecture decisions, security constraints, or immutable team conventions. ---- - -# Mem0 Pin - -Pin a memory to mark it as high-priority and protect from pruning. - -## Execution - -### Step 1: Find the memory - -The user provides either a search query or memory ID. - -**If memory ID:** -- Call `get_memory` with the ID. - -**If search query:** -- Call `search_memories` with the query, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=5`. -- Show numbered list with content previews. -- Ask: "Which memory to pin? Enter a number." - -### Step 2: Read current content - -Call `get_memory` with the selected memory ID. Store: -- `original_text` — the memory's text content -- `original_metadata` — the existing `metadata` dict - -### Step 3: Pin it - -The `update_memory` tool updates a memory by `id`. To pin durably, append a pin -marker to the text so it travels with the memory: - -```python -pinned_text = "[PINNED] " + original_text if not original_text.startswith("[PINNED]") else original_text -update_memory(id=, text=pinned_text) -``` - -**For new memories** (user wants to pin text that isn't stored yet): -1. Call `add_memory` with: - - `text="[PINNED] "` - - `user_id=` - - `app_id=` - - `metadata={"pinned": true, "type": "decision", "confidence": 1.0}` - - `infer=False` -2. The response contains `event_id`. Call `get_event_status(event_id=)` once to retrieve the memory ID, then confirm. - -### Step 4: Confirm - -``` -Pinned: "" -Memory ID: -``` - -Append `...` only if content exceeds 80 characters. - -### Unpin - -If the user says "unpin": -1. Call `get_memory` to read current content. -2. Remove the pin marker from the text: - ```python - unpinned_text = original_text.removeprefix("[PINNED] ") - update_memory(memory_id=, text=unpinned_text) - ``` -3. Print: `Unpinned: "..."` - -## Output formatting - -IMPORTANT: Do NOT use markdown in your output. OpenCode TUI renders text verbatim — markdown like **bold**, ## headers, and | table | syntax appears as raw characters. Use plain text with indentation for structure. Use dashes for lists. Use spaces to align columns instead of markdown tables. diff --git a/integrations/mem0-plugin/.opencode-plugin/project.ts b/integrations/mem0-plugin/.opencode-plugin/project.ts deleted file mode 100644 index cc89ffb6d..000000000 --- a/integrations/mem0-plugin/.opencode-plugin/project.ts +++ /dev/null @@ -1,20 +0,0 @@ -/** - * Project identity resolution for the Mem0 OpenCode plugin. - * - * The project id (`app_id`) scopes memories to a repo. We derive it from the - * git remote so it is stable across clones, worktrees, and sub-directories — - * falling back (in opencode-mem0.ts) to the git repo root dir name, then the - * cwd. Keeping the parser pure makes the tricky remote formats testable. - */ - -/** - * Parse `owner/repo` out of a git remote URL and return it as `owner-repo`. - * Handles https, scp-style ssh, custom ssh host aliases (e.g. - * `git@github.com-work:owner/repo.git`), an optional `.git` suffix, and a - * trailing slash. Returns null when no owner/repo can be found. - */ -export function parseProjectFromRemote(remote: string): string | null { - const m = remote.trim().match(/[:/]([^/:]+)\/([^/:]+?)(?:\.git)?\/?$/); - if (!m) return null; - return `${m[1]}-${m[2]}`; -} diff --git a/integrations/mem0-plugin/.opencode-plugin/scope.ts b/integrations/mem0-plugin/.opencode-plugin/scope.ts deleted file mode 100644 index 940f77767..000000000 --- a/integrations/mem0-plugin/.opencode-plugin/scope.ts +++ /dev/null @@ -1,71 +0,0 @@ -/** - * Memory scope resolution — ported from the pi-agent plugin's scoping model. - * - * Lets the agent choose, per memory operation, how wide to read/write: - * - "project" (default): this repo only -> { user_id, app_id } - * - "session": this run only -> { user_id, app_id, run_id } - * - "global": across ALL of the user's projects -> { user_id, app_id: "*" } - * - * Mirrors pi-agent/src/memory/scoping.ts (resolveSearchFilters / resolveAddParams) - * so the OpenCode plugin exposes scope the same way: as a per-call tool parameter, - * not a stateful "switch project" command. - */ - -export type Scope = "project" | "session" | "global"; - -/** Filters for `search` / `get_memories` at the given scope. */ -export function scopeSearchFilters( - scope: Scope, - userId: string, - appId: string, - runId: string, -): Record { - switch (scope) { - case "session": - return { user_id: userId, app_id: appId, run_id: runId }; - case "global": - return { user_id: userId, app_id: "*" }; - case "project": - default: - return { user_id: userId, app_id: appId }; - } -} - -/** Identity params for `add` / `delete_all` at the given scope. */ -export function scopeWriteParams( - scope: Scope, - userId: string, - appId: string, - runId: string, -): { user_id: string; app_id?: string; run_id?: string } { - switch (scope) { - case "session": - return { user_id: userId, app_id: appId, run_id: runId }; - case "global": - return { user_id: userId }; - case "project": - default: - return { user_id: userId, app_id: appId }; - } -} - -/** Normalize an arbitrary value to a valid Scope (defaults to "project"). */ -export function asScope(value: unknown): Scope { - return value === "session" || value === "global" ? value : "project"; -} - -/** - * Resolve the persisted default scope from a parsed `~/.mem0/settings.json` - * object. This is the user-changeable default applied to memory operations when - * no explicit `scope` is passed (set via the `mem0-scope` skill). Falls back to - * "project" when unset or invalid. - */ -export function resolveDefaultScope( - settings: Record | null | undefined, -): Scope { - return asScope(settings?.default_scope); -} - -/** Guidance injected so the agent uses `global` only when explicitly asked. */ -export const SCOPE_GUIDANCE = - 'Memory tools accept an optional `scope`: omit it (or "project") for normal queries; use "session" to limit to the current run; use "global" ONLY when the user explicitly asks to search across all their projects in this workspace.'; diff --git a/integrations/mem0-plugin/.opencode-plugin/telemetry.ts b/integrations/mem0-plugin/.opencode-plugin/telemetry.ts deleted file mode 100644 index f8557b21a..000000000 --- a/integrations/mem0-plugin/.opencode-plugin/telemetry.ts +++ /dev/null @@ -1,113 +0,0 @@ -/** - * Plugin telemetry for the Mem0 OpenCode plugin — anonymous usage tracking - * via PostHog. - * - * Emits the SAME event schema as the Mem0 editor plugin's telemetry.py - * (event names prefixed `plugin.`, `source: "plugin"`, `platform: "opencode"`, - * `distinct_id = sha256(apiKey)[:32]`) so OpenCode shows up as just another - * `platform` value in the shared plugin dashboard instead of a separate - * event namespace. - * - * Fire-and-forget: never throws, never blocks, failures are swallowed. Only - * fires when an API key is present (same as the editor plugin — anonymous - * installs without a key emit nothing). Disable with MEM0_TELEMETRY=false. - * - * Never sends: memory content, API keys, raw user/project IDs. Only sends: - * event type, platform, plugin version, and anonymized hashes of the API key - * and project ID. - */ - -import { createHash } from "node:crypto"; -import { readFileSync } from "node:fs"; -import { release } from "node:os"; - -const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"; -const POSTHOG_HOST = "https://us.i.posthog.com/i/v0/e/"; -const REQUEST_TIMEOUT_MS = 2_000; - -function _loadPluginVersion(): string { - // Source context: telemetry.ts sits next to package.json (./). - // Bundled context: dist/index.js sits one level below it (../). - for (const rel of ["./package.json", "../package.json"]) { - try { - const pkg = JSON.parse(readFileSync(new URL(rel, import.meta.url), "utf-8")); - if (pkg?.name === "@mem0/opencode-plugin" && pkg.version) return pkg.version; - } catch { - /* try next candidate */ - } - } - return "unknown"; -} - -const PLUGIN_VERSION = _loadPluginVersion(); - -export function isTelemetryEnabled(): boolean { - const val = process.env.MEM0_TELEMETRY; - if (val === undefined) return true; - const s = val.toLowerCase(); - return s !== "false" && s !== "0" && s !== "no" && s !== "off"; -} - -function distinctId(apiKey: string): string { - // Matches telemetry.py `_distinct_id()` so the same user is one person in - // PostHog whether they use OpenCode or any other Mem0 editor plugin. - return createHash("sha256").update(apiKey).digest("hex").slice(0, 32); -} - -/** - * Build the PostHog event payload, or null when telemetry is disabled or no - * API key is available. Pure (aside from env/version reads) and exported for - * testing. System-controlled properties are applied last so a caller cannot - * override `source`/`platform`/etc. - */ -export function buildEvent( - eventType: string, - properties: Record, - apiKey: string | undefined, - projectId?: string, -): Record | null { - if (!isTelemetryEnabled() || !apiKey) return null; - return { - api_key: POSTHOG_API_KEY, - distinct_id: distinctId(apiKey), - event: `plugin.${eventType}`, - properties: { - ...properties, - source: "plugin", - platform: "opencode", - plugin_version: PLUGIN_VERSION, - os: process.platform, - os_version: release(), - sample_rate: 1.0, - $process_person_profile: false, - $lib: "posthog-node", - // Anonymized project segmentation, matching telemetry.py's project_hash. - ...(projectId - ? { project_hash: createHash("sha256").update(projectId).digest("hex") } - : {}), - }, - }; -} - -/** Send a usage event, fire-and-forget. Never throws, never blocks. */ -export function captureEvent( - eventType: string, - properties: Record, - apiKey: string | undefined, - projectId?: string, -): void { - const payload = buildEvent(eventType, properties, apiKey, projectId); - if (!payload) return; - try { - void fetch(POSTHOG_HOST, { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(payload), - signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS), - }).catch(() => { - /* fire-and-forget */ - }); - } catch { - /* never throw */ - } -} diff --git a/integrations/mem0-plugin/README.md b/integrations/mem0-plugin/README.md deleted file mode 100644 index 1e381fbf4..000000000 --- a/integrations/mem0-plugin/README.md +++ /dev/null @@ -1,278 +0,0 @@ -> **Claude Code users:** version 0.3.0 of the Claude Code plugin now lives at -> [`integrations/claude-code-plugin`](../claude-code-plugin/) and is what `mem0@mem0-plugins` installs. -> Update with `claude plugin marketplace update mem0-plugins` then -> `claude plugin update mem0@mem0-plugins --scope user` — your memories carry over -> automatically. This directory continues to serve the Cursor, Codex, Kimi, -> Antigravity, and OpenCode integrations until they are ported. The Claude Code -> manifest, hooks, and MCP config have been removed from this directory, so -> there is nothing here left to install into Claude Code. - -# Mem0 Plugin for Cursor, Codex, Kimi, OpenCode & Antigravity - -Add persistent memory to your AI workflows. Store, retrieve, and manage memories across sessions using the Mem0 Platform. Works with **Cursor**, **Codex**, **Kimi**, **OpenCode**, and **Antigravity**. For Claude Code, use [`integrations/claude-code-plugin`](../claude-code-plugin/). - -## Quick path for agents - -If you're an AI agent setting up Mem0 autonomously (no human dashboard signup available), run: - -```bash -mem0 init --agent --json -``` - -This mints an evaluation API key in <5 seconds — no email, no OTP, no browser. The key is written to your config and exported as `MEM0_API_KEY` for the current shell. Use it immediately against the Mem0 Platform or the MCP server. The human owner can later run `mem0 init --email ` to claim ownership: memories transfer, the same API key keeps working, and the agent isn't disrupted. - -Humans setting up Mem0 by hand should continue with Step 1 below. - -## Step 1: Set your API key - -> **You must complete this step before installing the plugin.** - -1. Sign up at [app.mem0.ai](https://app.mem0.ai?utm_source=oss&utm_medium=mem0-plugin-readme) if you haven't already -2. Go to [app.mem0.ai/dashboard/api-keys](https://app.mem0.ai/dashboard/api-keys?utm_source=oss&utm_medium=mem0-plugin-readme) -3. Click **Create API Key** and copy the key (starts with `m0-`) -4. Set the key using **one** of these methods: - - **CLI** — add to your shell profile: - - ```bash - # For zsh (default on macOS) - echo 'export MEM0_API_KEY="m0-your-api-key"' >> ~/.zshrc - source ~/.zshrc - - # For bash - echo 'export MEM0_API_KEY="m0-your-api-key"' >> ~/.bashrc - source ~/.bashrc - ``` - - **Desktop app** — use the local environment editor: - - Click the environment dropdown next to the prompt box → hover over **Local** → click the **gear icon** → add `MEM0_API_KEY` with your key. Values are stored encrypted on your machine. - - > **Note:** The Desktop app does not inherit custom environment variables from shell profiles — it only reads `PATH`. You must use the local environment editor for Desktop. - -5. Confirm it's set: - - ```bash - echo $MEM0_API_KEY - # Should print: m0-your-api-key - ``` - -## Step 2: Install the plugin - -Choose one of the options below. All require `MEM0_API_KEY` to be set first (see above). - -### Codex - -**Option A — Direct MCP** (fastest, MCP only): - -Codex reads MCP servers from `~/.codex/config.toml` as TOML. Add: - -```toml -[mcp_servers.mem0] -url = "https://mcp.mem0.ai/mcp" -bearer_token_env_var = "MEM0_API_KEY" -``` - -Export `MEM0_API_KEY` in your shell and restart Codex. `codex mcp add` only supports stdio servers, so HTTP servers like Mem0's must be added via `config.toml` directly (or via the **Plugins → Connect to a custom MCP → Streamable HTTP** UI in the Codex app). - -**Option B — Sideload the plugin** (full experience: MCP + skills + opt-in hooks): - -Clone the repo and register the bundled marketplace with one CLI call: - -```bash -git clone https://github.com/mem0ai/mem0.git ~/codex-plugins/mem0-source -codex plugin marketplace add ~/codex-plugins/mem0-source -``` - -This points Codex at the repo's `.agents/plugins/marketplace.json`, which references `integrations/mem0-plugin/` as the local source. Restart Codex, run `/plugins`, and install **Mem0** from the **Mem0 Plugins** marketplace. - -> **Don't combine with Option A.** The plugin manifest auto-registers `mem0` as an MCP server via `integrations/mem0-plugin/.codex-mcp.json` — adding a manual `[mcp_servers.mem0]` block would duplicate the registration. - -**Optional — enable lifecycle hooks.** Codex doesn't auto-wire hooks from plugin manifests; it only reads `~/.codex/hooks.json` (or `/.codex/hooks.json`) ([docs](https://developers.openai.com/codex/hooks)). Run the bundled installer once to merge Mem0's entries: - -```bash -python3 ~/codex-plugins/mem0-source/integrations/mem0-plugin/scripts/install_codex_hooks.py -``` - -This merges six event handlers into `~/.codex/hooks.json` with absolute paths pointing into your clone: - -| Event | What it does | -|-------|--------------| -| `SessionStart` | Loads prior memories as bootstrap context | -| `UserPromptSubmit` | Injects relevant memories into the prompt | -| `PreToolUse` (3 handlers) | Blocks MEMORY.md writes; enforces `user_id`/`app_id` on mem0 tool calls; scans files being read for relevant memory context | -| `PostToolUse` (2 handlers) | Tracks stats, scans bash errors for related memories | -| `Stop` | Reminds the agent to persist learnings at turn end | -| `PreCompact` | Stores a summary before the context is compacted | - -Re-running the installer is idempotent (replaces the Mem0 entries rather than duplicating) and preserves any other hooks you have. To remove: `python3 .../install_codex_hooks.py --uninstall`. If you move or delete the clone directory, re-run the installer from the new location — the hooks file stores absolute paths. - -Codex hooks also require the `codex_hooks` feature flag in `~/.codex/config.toml`: - -```toml -[features] -codex_hooks = true -``` - -The installer prints a reminder if the flag isn't set. Restart Codex after editing the config. - -**Managing the plugin:** - -```bash -codex plugin marketplace upgrade # pull latest plugin versions -codex plugin marketplace remove mem0-plugins # unregister the marketplace -``` - -### Cursor - -> **Already have `mem0` configured as an MCP server?** Remove the existing entry from your Cursor MCP settings before installing to avoid duplicate tools. - -**Option A — One-click deeplink** (installs MCP server only): - -[Install Mem0 MCP in Cursor](cursor://anysphere.cursor-deeplink/mcp/install?name=mem0&config=eyJtY3BTZXJ2ZXJzIjp7Im1lbTAiOnsidXJsIjoiaHR0cHM6Ly9tY3AubWVtMC5haS9tY3AvIiwiaGVhZGVycyI6eyJBdXRob3JpemF0aW9uIjoiVG9rZW4gJHtlbnY6TUVNMF9BUElfS0VZfSJ9fX19) - -**Option B — Manual configuration** (MCP server only): - -Add the following to your `.cursor/mcp.json`: - -```json -{ - "mcpServers": { - "mem0": { - "url": "https://mcp.mem0.ai/mcp/", - "headers": { - "Authorization": "Token ${env:MEM0_API_KEY}" - } - } - } -} -``` - -### OpenCode - -```bash -opencode plugin @mem0/opencode-plugin -``` - -Add `--global` to install for all projects. The plugin auto-registers its native memory tools, hooks, and skills via its `config` hook — no MCP server to configure. Restart OpenCode after installing. - -See [OpenCode integration docs](https://docs.mem0.ai/integrations/opencode) for full details. - -### Antigravity (Google) - -**Option A — degit** (recommended): - -```bash -# Install the plugin (MCP server, hooks, scripts) -npx degit mem0ai/mem0/integrations/mem0-plugin ~/.gemini/config/plugins/mem0 -``` - -This installs the MCP server, lifecycle hooks, and shared scripts. - -See [Antigravity integration docs](https://docs.mem0.ai/integrations/antigravity) for full details. - -## Post-Installation: Run `/mem0:onboard` - -After installing, start a new session and run: - -``` -/mem0:onboard -``` - -This runs the setup wizard which: -1. Verifies your API key and MCP connection -2. Detects and imports project files (`CLAUDE.md`, `AGENTS.md`, `.cursorrules`) -3. Installs coding-optimized memory categories -4. Shows your identity (user ID, project scope, branch) - -The onboarding is idempotent — safe to re-run anytime. On first session in a new project (0 memories), the agent is prompted to run it automatically. - -## Verify it works - -After onboarding, confirm everything is connected: - -1. Run `/mem0:health` to check connectivity -2. Run `/mem0:stats` to see memory counts -3. Try `/mem0:remember "we use TypeScript"` then `/mem0:tour` to see it stored - -## Available Skills - -The plugin includes 17 skills accessible via `/mem0:` commands: - -| Command | Description | -|---------|-------------| -| `/mem0:remember` | Store a memory verbatim — decisions, preferences, conventions | -| `/mem0:tour` | Browse all memories grouped by category | -| `/mem0:peek` | Quick search with compact one-liner results | -| `/mem0:stats` | Session and project memory statistics | -| `/mem0:dream` | Consolidate memories — merge duplicates, resolve contradictions | -| `/mem0:pin` | Protect critical memories from pruning | -| `/mem0:forget` | Delete memories by search or ID | -| `/mem0:health` | Diagnose connectivity, API key, and read/write | -| `/mem0:export` | Export memories to portable Markdown | -| `/mem0:import` | Import memories from export file or MEMORY.md | -| `/mem0:list-projects` | List all projects with stored memories | -| `/mem0:switch-project` | Override auto-detected project scope | -| `/mem0:memory-reviewer` | Audit memory quality — duplicates, contradictions, stale | -| `/mem0:context-loader` | Pre-load relevant memories for current task | - -## What's included - -| Component | Cursor (MCP) | Codex (Sideload) | Codex (Direct MCP) | OpenCode (Full) | OpenCode (MCP) | Antigravity | -|-----------|:------------:|:----------------:|:------------------:|:---------------:|:--------------:|:-----------:| -| MCP Server | Yes | Yes | Yes | Yes | Yes | Yes | -| Lifecycle Hooks | No | Opt-in | No | Yes | No | Yes | -| Mem0 SDK Skill | No | Yes | No | Yes | No | Yes | - -- **MCP Server** — Connects to the Mem0 remote MCP server (`mcp.mem0.ai`), providing tools to add, search, update, and delete memories. No local dependencies required. -- **Lifecycle Hooks** — Automatic memory capture at key points. OpenCode and Antigravity wire hooks natively when the full plugin is installed. Codex hooks are opt-in via a one-time installer (`scripts/install_codex_hooks.py`). -- **Mem0 SDK Skill** — Guides the AI on how to integrate the Mem0 SDK (Python & TypeScript) into your applications. - -## Updating the plugin - -When the plugin updates (new version pulled from the marketplace, or a fresh local install), the MCP server connection in your existing session is left holding a stale handle and stops responding. **Restart your client to reconnect:** - -- **Cursor:** quit and relaunch. -- **Codex:** restart the editor session. -- **OpenCode:** restart the session. -- **Antigravity:** restart the session. - -Your `MEM0_API_KEY` doesn't need to be re-entered — the auth header is re-read from your environment on the new session. The plugin's MCP config uses `${MEM0_API_KEY}` interpolation at session start, not at install time, so as long as the env var is set persistently (in your shell profile), reconnection is automatic on restart. - -If reconnection still fails after a restart, check that `MEM0_API_KEY` is reachable in the new shell (`echo $MEM0_API_KEY`) and confirm you're using a key that starts with `m0-` (from https://app.mem0.ai/dashboard/api-keys, not a legacy token). - -## Coding-tuned categories (automatic) - -mem0 auto-tags every memory with one or more `categories` from a project-level list. The default list is consumer-oriented (`food`, `hobbies`, `music` …) — useful for chat assistants, less so for code. **The plugin installs a coding-focused taxonomy automatically in the background on session start** — no prompt, no manual step. New memories then auto-tag against 17 development-oriented categories: `architecture_decisions`, `anti_patterns`, `task_learnings`, `tooling_setup`, `bug_fixes`, `coding_conventions`, `user_preferences`, `dependency_decisions`, `performance_findings`, `security_constraints`, `testing_patterns`, `data_model`, `api_contracts`, `deployment_runbook`, `team_norms`, `domain_glossary`, `experiment_results`. - -The background setup is idempotent and runs once per account (cached in `~/.mem0/categories_setup.json`); it re-applies only if the taxonomy itself changes. To preview the taxonomy or force a refresh manually: - -```bash -# Dry-run -- prints current vs proposed, no changes: -python integrations/mem0-plugin/scripts/setup_coding_categories.py - -# Write explicitly: -python integrations/mem0-plugin/scripts/setup_coding_categories.py --apply -``` - -Requires the `mem0ai` Python SDK (`pip install mem0ai`) and `MEM0_API_KEY` set. `project.update(custom_categories=[...])` always replaces the full list. - -## MCP Tools - -Once installed, the following tools are available: - -| Tool | Description | -|------|-------------| -| `add_memory` | Save text or conversation history for a user/agent | -| `search_memories` | Semantic search across memories with filters | -| `get_memories` | List memories with filters and pagination | -| `get_memory` | Retrieve a specific memory by ID | -| `update_memory` | Overwrite a memory's text by ID | -| `delete_memory` | Delete a single memory by ID | -| `delete_all_memories` | Bulk delete all memories in scope | -| `delete_entities` | Delete a user/agent/app/run entity and its memories | -| `list_entities` | List users/agents/apps/runs stored in Mem0 | - -## License - -Apache-2.0 diff --git a/integrations/mem0-plugin/hooks.json b/integrations/mem0-plugin/hooks.json deleted file mode 100644 index c7b9144b1..000000000 --- a/integrations/mem0-plugin/hooks.json +++ /dev/null @@ -1,104 +0,0 @@ -{ - "hooks": { - "SessionStart": [ - { - "matcher": "*", - "hooks": [ - { - "name": "mem0-ensure-deps", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/ensure_deps.sh || true", - "timeout": 60 - }, - { - "name": "mem0-session-start", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/on_session_start.sh || true" - } - ] - } - ], - "UserPromptSubmit": [ - { - "hooks": [ - { - "name": "mem0-user-prompt", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/on_user_prompt.sh || true", - "timeout": 8 - } - ] - } - ], - "PreToolUse": [ - { - "matcher": "Write|Edit|MultiEdit", - "hooks": [ - { - "name": "mem0-block-memory-write", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/block_memory_write.sh" - } - ] - }, - { - "matcher": "mcp__mem0__.*|mcp__plugin_mem0_mem0__.*", - "hooks": [ - { - "name": "mem0-enforce-metadata", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/enforce_metadata_defaults.sh", - "timeout": 3 - } - ] - }, - { - "matcher": "Read", - "hooks": [ - { - "name": "mem0-file-context", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/on_file_read.sh", - "timeout": 5 - } - ] - } - ], - "Stop": [ - { - "hooks": [ - { - "name": "mem0-session-summary", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/on_stop.sh", - "timeout": 30 - } - ] - } - ], - "PostToolUse": [ - { - "matcher": "mcp__mem0__.*|mcp__plugin_mem0_mem0__.*", - "hooks": [ - { - "name": "mem0-post-tool", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/on_post_tool_use.sh", - "timeout": 3 - } - ] - }, - { - "matcher": "Bash", - "hooks": [ - { - "name": "mem0-bash-output", - "type": "command", - "command": "ANTIGRAVITY_PLUGIN_ROOT=${extensionPath} CLAUDE_PLUGIN_ROOT=${extensionPath} bash ${extensionPath}/scripts/on_bash_output.sh", - "timeout": 5 - } - ] - } - ] - } -} diff --git a/integrations/mem0-plugin/hooks/codex-hooks.json b/integrations/mem0-plugin/hooks/codex-hooks.json deleted file mode 100644 index 7e0cb002f..000000000 --- a/integrations/mem0-plugin/hooks/codex-hooks.json +++ /dev/null @@ -1,104 +0,0 @@ -{ - "hooks": { - "PreToolUse": [ - { - "matcher": "Write|Edit|MultiEdit", - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/block_memory_write.sh", - "timeout": 3 - } - ] - }, - { - "matcher": "mcp__mem0__add_memory|mcp__plugin_mem0_mem0__add_memory|mcp__mem0__search_memories|mcp__plugin_mem0_mem0__search_memories|mcp__mem0__get_memories|mcp__plugin_mem0_mem0__get_memories|mcp__mem0__delete_all_memories|mcp__plugin_mem0_mem0__delete_all_memories", - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/enforce_metadata_defaults.sh", - "timeout": 3 - } - ] - }, - { - "matcher": "Read", - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/on_file_read.sh", - "timeout": 5 - } - ] - } - ], - "SessionStart": [ - { - "matcher": "startup|resume|compact", - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/on_session_start.sh", - "statusMessage": "Loading mem0 context..." - } - ] - } - ], - "UserPromptSubmit": [ - { - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/on_user_prompt.sh", - "statusMessage": "Checking memory relevance...", - "timeout": 12 - } - ] - } - ], - "PostToolUse": [ - { - "matcher": "mcp__mem0__.*|mcp__plugin_mem0_mem0__.*", - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/on_post_tool_use.sh", - "timeout": 3 - } - ] - }, - { - "matcher": "Bash", - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/on_bash_output.sh", - "timeout": 12 - } - ] - } - ], - "Stop": [ - { - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/on_stop.sh", - "timeout": 30 - } - ] - } - ], - "PreCompact": [ - { - "hooks": [ - { - "type": "command", - "command": "MEM0_PLATFORM=codex ${PLUGIN_ROOT}/scripts/on_pre_compact.sh", - "statusMessage": "Preparing pre-compaction summary..." - } - ] - } - ] - } -} diff --git a/integrations/mem0-plugin/hooks/cursor-hooks.json b/integrations/mem0-plugin/hooks/cursor-hooks.json deleted file mode 100644 index 1a07cb23b..000000000 --- a/integrations/mem0-plugin/hooks/cursor-hooks.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "version": 1, - "hooks": { - "sessionStart": [ - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_session_start_cursor.sh", - "matcher": "startup|resume|compact" - } - ], - "preToolUse": [ - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/block_memory_write_cursor.sh", - "matcher": "Write|Edit|MultiEdit" - }, - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/enforce_metadata_defaults.sh", - "matcher": "mcp__mem0__add_memory|mcp__plugin_mem0_mem0__add_memory|mcp__mem0__search_memories|mcp__plugin_mem0_mem0__search_memories|mcp__mem0__get_memories|mcp__plugin_mem0_mem0__get_memories|mcp__mem0__delete_all_memories|mcp__plugin_mem0_mem0__delete_all_memories", - "timeout": 3 - }, - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_file_read_cursor.sh", - "matcher": "Read", - "timeout": 5 - } - ], - "postToolUse": [ - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_post_tool_use_cursor.sh", - "matcher": "mcp__mem0__.*|mcp__plugin_mem0_mem0__.*", - "timeout": 3 - }, - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_bash_output.sh", - "matcher": "Bash", - "timeout": 12 - } - ], - "stop": [ - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_stop_cursor.sh", - "timeout": 30 - } - ], - "preCompact": [ - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_pre_compact_cursor.sh" - } - ], - "beforeSubmitPrompt": [ - { - "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_user_prompt_cursor.sh", - "timeout": 12 - } - ] - } -} diff --git a/integrations/mem0-plugin/logo.svg b/integrations/mem0-plugin/logo.svg deleted file mode 100644 index cf13d0d38..000000000 --- a/integrations/mem0-plugin/logo.svg +++ /dev/null @@ -1,19 +0,0 @@ - - - - - - - - - - - - - - - - - - - diff --git a/integrations/mem0-plugin/mcp_config.json b/integrations/mem0-plugin/mcp_config.json deleted file mode 100644 index 86e2428d5..000000000 --- a/integrations/mem0-plugin/mcp_config.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "mcpServers": { - "mem0": { - "serverUrl": "https://mcp.mem0.ai/mcp/", - "headers": { - "Authorization": "Token ${MEM0_API_KEY}" - } - } - } -} diff --git a/integrations/mem0-plugin/plugin.json b/integrations/mem0-plugin/plugin.json deleted file mode 100644 index 35d1fdf63..000000000 --- a/integrations/mem0-plugin/plugin.json +++ /dev/null @@ -1,13 +0,0 @@ -{ - "id": "mem0", - "name": "mem0", - "version": "0.1.7", - "description": "Persistent semantic memory for Antigravity agents. Cross-session, user-level recall via the Mem0 Platform MCP server. 16 slash commands, lifecycle hooks for auto-capture and metadata enforcement.", - "author": { "name": "Mem0", "email": "support@mem0.ai" }, - "publisher": "mem0ai", - "homepage": "https://mem0.ai", - "repository": "https://github.com/mem0ai/mem0", - "license": "Apache-2.0", - "keywords": ["memory", "persistence", "personalization", "mcp", "semantic-search"], - "contextFileName": "AGENTS.md" -} diff --git a/integrations/mem0-plugin/requirements.txt b/integrations/mem0-plugin/requirements.txt deleted file mode 100644 index 633d7eaad..000000000 --- a/integrations/mem0-plugin/requirements.txt +++ /dev/null @@ -1 +0,0 @@ -mem0ai diff --git a/integrations/mem0-plugin/scripts/_chunking.py b/integrations/mem0-plugin/scripts/_chunking.py deleted file mode 100644 index 359199d9d..000000000 --- a/integrations/mem0-plugin/scripts/_chunking.py +++ /dev/null @@ -1,75 +0,0 @@ -"""Shared content-chunking utilities for mem0-plugin import scripts.""" - -from __future__ import annotations - -MIN_CHUNK_CHARS = 50 -MAX_CHUNK_CHARS = 10_000 - - -def split_by_headers(content: str, header_prefix: str = "## ") -> list[str]: - """Split content by Markdown header lines (e.g. '## '). - - The header line is included at the start of each chunk. - Returns a list of non-empty chunk strings. - """ - chunks: list[str] = [] - current_lines: list[str] = [] - - for line in content.splitlines(keepends=True): - if line.startswith(header_prefix) and current_lines: - chunk = "".join(current_lines).strip() - if chunk: - chunks.append(chunk) - current_lines = [line] - else: - current_lines.append(line) - - if current_lines: - chunk = "".join(current_lines).strip() - if chunk: - chunks.append(chunk) - - return chunks - - -def split_by_hr_or_headers(content: str) -> list[str]: - """Split content by '---' horizontal rules or '## ' headers. - - Used for .continue/rules.md which may use either convention. - """ - import re - - # Split on lines that are exactly "---" or start with "## " - chunks: list[str] = [] - current_lines: list[str] = [] - - for line in content.splitlines(keepends=True): - is_hr = re.match(r"^---\s*$", line) - is_h2 = line.startswith("## ") - - if (is_hr or is_h2) and current_lines: - chunk = "".join(current_lines).strip() - if chunk: - chunks.append(chunk) - current_lines = [] if is_hr else [line] - else: - current_lines.append(line) - - if current_lines: - chunk = "".join(current_lines).strip() - if chunk: - chunks.append(chunk) - - return chunks - - -def filter_and_truncate(chunks: list[str]) -> list[str]: - """Filter out chunks shorter than MIN_CHUNK_CHARS, truncate long chunks.""" - result: list[str] = [] - for chunk in chunks: - if len(chunk) < MIN_CHUNK_CHARS: - continue - if len(chunk) > MAX_CHUNK_CHARS: - chunk = chunk[:MAX_CHUNK_CHARS] - result.append(chunk) - return result diff --git a/integrations/mem0-plugin/scripts/_formatting.py b/integrations/mem0-plugin/scripts/_formatting.py deleted file mode 100644 index eb5e23368..000000000 --- a/integrations/mem0-plugin/scripts/_formatting.py +++ /dev/null @@ -1,51 +0,0 @@ -"""Shared formatting helpers for mem0 plugin hooks. - -Constants and utilities used by file_context.py, session_timeline.py, -and any future hook that displays memories. -""" - -from __future__ import annotations - -TYPE_ICONS = { - "decision": "⚖️", - "anti_pattern": "\U0001f534", - "bug_fix": "\U0001f534", - "convention": "\U0001f504", - "task_learning": "\U0001f535", - "user_preference": "\U0001f7e3", - "session_summary": "\U0001f4cb", - "session_state": "\U0001f4cb", - "project_profile": "\U0001f4d6", - "compact_summary": "\U0001f4cb", - "auto_capture": "✅", - "environmental": "🌐", - "health_check": "🩺", -} - - -def format_age(memory: dict) -> str: - """Format how long ago a memory was created, e.g. '2h ago', '3d ago'.""" - created = memory.get("created_at", "") - if not created: - return "" - try: - from datetime import datetime, timezone - - if created.endswith("Z"): - created = created[:-1] + "+00:00" - dt = datetime.fromisoformat(created) - now = datetime.now(timezone.utc) - delta = now - dt - seconds = int(delta.total_seconds()) - if seconds < 3600: - return f"{seconds // 60}m ago" - if seconds < 86400: - return f"{seconds // 3600}h ago" - days = seconds // 86400 - if days == 1: - return "1d ago" - if days < 30: - return f"{days}d ago" - return f"{days // 30}mo ago" - except Exception: - return "" diff --git a/integrations/mem0-plugin/scripts/_identity.py b/integrations/mem0-plugin/scripts/_identity.py deleted file mode 100644 index d60ccb4a3..000000000 --- a/integrations/mem0-plugin/scripts/_identity.py +++ /dev/null @@ -1,104 +0,0 @@ -"""Resolve mem0 identity: API key, user_id, and settings. - -API key resolution (first non-empty wins): - 1. MEM0_API_KEY env var (explicit / shell profile) - 2. CLAUDE_PLUGIN_OPTION_API_KEY (injected by Claude Code userConfig) - 3. CLAUDE_PLUGIN_OPTION_MEM0_API_KEY (legacy userConfig) - 4. Extract from shell profile files (~/.zshrc, ~/.bashrc, etc.) - Desktop app doesn't inherit shell env — this covers users who - set MEM0_API_KEY in their profile but use the Desktop app. - -User ID resolution: - 1. MEM0_USER_ID env var (explicit override) - 2. $USER, else "default" - -Settings resolution: - ~/.mem0/settings.json (user-editable, falls back to defaults) -""" - -from __future__ import annotations - -import os -import re -from pathlib import Path - - -def _extract_key_from_shell_profiles() -> str: - """Extract MEM0_API_KEY from shell profile files. - - The Desktop app only reads PATH from shell profiles — env vars like - MEM0_API_KEY are not inherited. This handles the common - ``export MEM0_API_KEY=...`` pattern without sourcing the full profile. - """ - profiles = [".zshrc", ".bashrc", ".zprofile", ".bash_profile", ".profile"] - pattern = re.compile(r'^\s*(?:export\s+)?MEM0_API_KEY=(.+)$') - - for name in profiles: - path = Path.home() / name - if not path.is_file(): - continue - try: - for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): - m = pattern.match(line) - if not m: - continue - value = m.group(1).strip() - value = re.sub(r'#.*$', '', value).strip() - value = value.strip("\"'") - if value and not value.startswith("$"): - return value - except OSError: - continue - return "" - - -def resolve_api_key() -> str: - key = os.environ.get("MEM0_API_KEY", "").strip() - if key: - return key - key = os.environ.get("CLAUDE_PLUGIN_OPTION_API_KEY", "").strip() - if key: - return key - key = os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", "").strip() - if key: - return key - key = _extract_key_from_shell_profiles() - if key: - return key - return "" - - -def resolve_user_id() -> str: - explicit = os.environ.get("MEM0_USER_ID", "").strip() - if explicit: - return explicit - return os.environ.get("USER") or "default" - - -def resolve_config() -> dict: - """Resolve settings from ~/.mem0/settings.json (primary) with env var overrides.""" - try: - from load_settings import load_settings - return load_settings() - except ImportError: - return { - "auto_save": True, - "auto_search": True, - "search_limit": 10, - "retention_session_days": 90, - "confidence_threshold": 0.3, - "debug": False, - } - - -try: - from _project import resolve_branch, resolve_project_id, save_project_mapping -except ImportError: - def resolve_project_id(cwd: str | None = None) -> str: - return os.path.basename(cwd or os.getcwd()) - - def resolve_branch(cwd: str | None = None) -> str: - return "unknown" - - def save_project_mapping(cwd: str, project_id: str) -> None: - pass diff --git a/integrations/mem0-plugin/scripts/_identity.sh b/integrations/mem0-plugin/scripts/_identity.sh deleted file mode 100644 index 4a25b8afc..000000000 --- a/integrations/mem0-plugin/scripts/_identity.sh +++ /dev/null @@ -1,87 +0,0 @@ -# Source this file. Sets MEM0_API_KEY, MEM0_RESOLVED_USER_ID, and settings. -# -# API key resolution (first non-empty wins): -# 1. MEM0_API_KEY env var (explicit / shell profile) -# 2. CLAUDE_PLUGIN_OPTION_API_KEY (injected by Claude Code userConfig) -# 3. CLAUDE_PLUGIN_OPTION_MEM0_API_KEY (legacy userConfig) -# 4. Extract from shell profile files (~/.zshrc, ~/.bashrc, etc.) -# Desktop app doesn't inherit shell env — this fallback covers users -# who set MEM0_API_KEY in their profile but use the Desktop app. -# -# Settings: ~/.mem0/settings.json (user-editable, falls back to defaults) - -_SCRIPT_DIR="$( cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd )" - -# Resolve API key: env var > userConfig > shell profile extraction -if [ -z "${MEM0_API_KEY:-}" ] && [ -n "${CLAUDE_PLUGIN_OPTION_API_KEY:-}" ]; then - MEM0_API_KEY="$CLAUDE_PLUGIN_OPTION_API_KEY" - export MEM0_API_KEY -fi -if [ -z "${MEM0_API_KEY:-}" ] && [ -n "${CLAUDE_PLUGIN_OPTION_MEM0_API_KEY:-}" ]; then - MEM0_API_KEY="$CLAUDE_PLUGIN_OPTION_MEM0_API_KEY" - export MEM0_API_KEY -fi -# Fallback: extract MEM0_API_KEY from shell profile files. -# The Desktop app only reads PATH from shell profiles — env vars like -# MEM0_API_KEY are not inherited. This grep-based extraction handles -# the common `export MEM0_API_KEY=...` pattern without sourcing the -# full profile (which could have side effects). -if [ -z "${MEM0_API_KEY:-}" ]; then - for _profile in "$HOME/.zshrc" "$HOME/.bashrc" "$HOME/.zprofile" "$HOME/.bash_profile" "$HOME/.profile"; do - if [ -f "$_profile" ]; then - _extracted=$(grep -E '^\s*(export\s+)?MEM0_API_KEY=' "$_profile" 2>/dev/null \ - | tail -1 \ - | sed 's/^[^=]*=//' \ - | sed "s/^[\"']//;s/[\"']$//" \ - | sed 's/#.*//' \ - | tr -d '[:space:]') - # Only use literal values — skip variable references like ${OTHER_VAR} - if [ -n "$_extracted" ] && [ "${_extracted#\$}" = "$_extracted" ]; then - MEM0_API_KEY="$_extracted" - export MEM0_API_KEY - break - fi - fi - done -fi - -_mem0_resolve_identity() { - if [ -n "${MEM0_USER_ID:-}" ]; then - printf '%s' "$MEM0_USER_ID" - return - fi - printf '%s' "${USER:-default}" -} - -MEM0_RESOLVED_USER_ID="$(_mem0_resolve_identity)" -export MEM0_RESOLVED_USER_ID - -_MEM0_IDENTITY_ANNOTATION="" -if [ -n "${MEM0_USER_ID:-}" ] && [ "$MEM0_USER_ID" != "${USER:-default}" ]; then - _MEM0_IDENTITY_ANNOTATION=" (override; default: ${USER:-default})" -fi -export _MEM0_IDENTITY_ANNOTATION - -# Load settings from ~/.mem0/settings.json -if command -v python3 >/dev/null 2>&1; then - _SETTINGS_JSON=$(PYTHONPATH="$_SCRIPT_DIR" python3 -c "from load_settings import load_settings; import json; print(json.dumps(load_settings()))" 2>/dev/null || echo "{}") - MEM0_AUTO_SAVE=$(echo "$_SETTINGS_JSON" | python3 -c "import sys,json; print(str(json.load(sys.stdin).get('auto_save',True)).lower())" 2>/dev/null || echo "true") - MEM0_AUTO_SEARCH=$(echo "$_SETTINGS_JSON" | python3 -c "import sys,json; print(str(json.load(sys.stdin).get('auto_search',True)).lower())" 2>/dev/null || echo "true") - MEM0_SEARCH_LIMIT=$(echo "$_SETTINGS_JSON" | python3 -c "import sys,json; print(json.load(sys.stdin).get('search_limit',10))" 2>/dev/null || echo "10") - MEM0_RETENTION_SESSION_DAYS=$(echo "$_SETTINGS_JSON" | python3 -c "import sys,json; print(json.load(sys.stdin).get('retention_session_days',90))" 2>/dev/null || echo "90") - MEM0_CONFIDENCE_THRESHOLD=$(echo "$_SETTINGS_JSON" | python3 -c "import sys,json; print(json.load(sys.stdin).get('confidence_threshold',0.3))" 2>/dev/null || echo "0.3") - MEM0_GLOBAL_SEARCH=$(echo "$_SETTINGS_JSON" | python3 -c "import sys,json; print(str(json.load(sys.stdin).get('global_search',False)).lower())" 2>/dev/null || echo "false") - MEM0_DEBUG=$(echo "$_SETTINGS_JSON" | python3 -c "import sys,json; print(str(json.load(sys.stdin).get('debug',False)).lower())" 2>/dev/null || echo "false") -else - MEM0_AUTO_SAVE="true" - MEM0_AUTO_SEARCH="true" - MEM0_SEARCH_LIMIT="10" - MEM0_RETENTION_SESSION_DAYS="90" - MEM0_CONFIDENCE_THRESHOLD="0.3" - MEM0_GLOBAL_SEARCH="false" - MEM0_DEBUG="false" -fi -export MEM0_AUTO_SAVE MEM0_AUTO_SEARCH MEM0_SEARCH_LIMIT MEM0_RETENTION_SESSION_DAYS MEM0_CONFIDENCE_THRESHOLD MEM0_GLOBAL_SEARCH MEM0_DEBUG - -# Also resolve project context -. "$_SCRIPT_DIR/_project.sh" diff --git a/integrations/mem0-plugin/scripts/_instructions.py b/integrations/mem0-plugin/scripts/_instructions.py deleted file mode 100644 index 8b4a5e8ae..000000000 --- a/integrations/mem0-plugin/scripts/_instructions.py +++ /dev/null @@ -1,51 +0,0 @@ -"""Resolve the project's mem0 extraction policy from ``mem0.md``. - -A repo's ``mem0.md`` can carry two prose sections that steer what Mem0 extracts: - - ## Instructions - Remember architecture decisions and conventions. Ignore debug noise and secrets. - - ## Agent Instructions - For agent-scoped memories, focus on the tools and task outcomes. - -``## Instructions`` maps to Mem0's ``custom_instructions`` (user/project-scoped -extraction) and ``## Agent Instructions`` to ``agent_custom_instructions`` -(agent-scoped extraction). Both are passed verbatim on memory writes, so the -policy lives in the repo, travels with it, and is shared by the whole team. - -The hook writers call :func:`load_instructions` and merge the result into their -``/v3/memories/add`` body. Returns only the keys that are actually set, so a -project with no policy adds nothing. -""" - -from __future__ import annotations - -import os - -from parse_mem0_config import load_full_config - - -def load_instructions(cwd: str | None = None) -> dict[str, str]: - """Return the extraction policy for the project at *cwd* (defaults to the - ``MEM0_CWD`` env var, then the process cwd). - - Keys (present only when non-empty): - - ``custom_instructions`` from ``## Instructions`` - - ``agent_custom_instructions`` from ``## Agent Instructions`` - """ - if cwd is None: - cwd = os.environ.get("MEM0_CWD") or os.getcwd() - - try: - config = load_full_config(cwd) - except Exception: - return {} - - out: dict[str, str] = {} - custom = config.get("instructions") - if isinstance(custom, str) and custom.strip(): - out["custom_instructions"] = custom.strip() - agent = config.get("agent_instructions") - if isinstance(agent, str) and agent.strip(): - out["agent_custom_instructions"] = agent.strip() - return out diff --git a/integrations/mem0-plugin/scripts/_project.py b/integrations/mem0-plugin/scripts/_project.py deleted file mode 100644 index eb64272b5..000000000 --- a/integrations/mem0-plugin/scripts/_project.py +++ /dev/null @@ -1,175 +0,0 @@ -"""Resolve mem0 project_id and branch. - -Resolution priority (project_id): - 1. MEM0_PROJECT_ID env var (explicit override) - 2. ~/.mem0/project_map.json lookup by cwd - 2b. ~/.mem0/project_map.json lookup by remote hash (self-healing fallback) - 3. Git remote slug: strip protocol/prefix, strip .git, replace / and : with - - e.g. git@github.com:mem0ai/mem0.git -> mem0ai-mem0 - 4. Fallback: basename of cwd -""" - -from __future__ import annotations - -import hashlib -import json -import os -import re -import subprocess - - -def resolve_project_id(cwd: str | None = None) -> str: - if cwd is None: - cwd = os.getcwd() - - # 1. Explicit override - explicit = os.environ.get("MEM0_PROJECT_ID", "").strip() - if explicit: - return explicit - - # 2. project_map.json lookup - map_path = os.path.expanduser("~/.mem0/project_map.json") - if os.path.isfile(map_path): - try: - with open(map_path) as f: - project_map = json.load(f) - mapped = project_map.get(cwd, "").strip() - if mapped: - return mapped - # 2b. Remote hash fallback (self-healing when folder is moved/renamed) - remote_key = _remote_hash_key(cwd) - if remote_key: - mapped = project_map.get(remote_key, "").strip() - if mapped: - # Self-heal: write the new CWD key so future lookups are fast - project_map[cwd] = mapped - try: - with open(map_path, "w") as f: - json.dump(project_map, f, indent=2) - except OSError: - pass - return mapped - except (OSError, json.JSONDecodeError, AttributeError): - pass - - # 3. Git remote slug - try: - result = subprocess.run( - ["git", "remote", "get-url", "origin"], - capture_output=True, - text=True, - check=True, - cwd=cwd, - ) - remote_url = result.stdout.strip() - if remote_url: - slug = _remote_url_to_slug(remote_url) - if slug: - return slug - except (subprocess.CalledProcessError, OSError): - pass - - # 4. Fallback: basename of cwd - return os.path.basename(cwd) or "unknown" - - -def resolve_branch(cwd: str | None = None) -> str: - if cwd is None: - cwd = os.getcwd() - try: - result = subprocess.run( - ["git", "branch", "--show-current"], - capture_output=True, - text=True, - check=True, - cwd=cwd, - ) - branch = result.stdout.strip() - return branch if branch else "unknown" - except (subprocess.CalledProcessError, OSError): - return "unknown" - - -def save_project_mapping(cwd: str, project_id: str) -> None: - """Write cwd -> project_id (and remote hash key -> project_id) into ~/.mem0/project_map.json.""" - mem0_dir = os.path.expanduser("~/.mem0") - os.makedirs(mem0_dir, exist_ok=True) - map_path = os.path.join(mem0_dir, "project_map.json") - project_map: dict[str, str] = {} - if os.path.isfile(map_path): - try: - with open(map_path) as f: - project_map = json.load(f) - except (OSError, json.JSONDecodeError): - project_map = {} - project_map[cwd] = project_id - # Also write the remote hash key so the mapping survives folder moves/renames - remote_key = _remote_hash_key(cwd) - if remote_key: - project_map[remote_key] = project_id - with open(map_path, "w") as f: - json.dump(project_map, f, indent=2) - - -def _remote_hash_key(cwd: str | None = None) -> str: - """Return a stable key derived from the git remote URL. - - Runs ``git config --get remote.origin.url`` in *cwd* and returns a string - of the form ``remote:``. Returns an empty string when - the directory is not a git repo or has no remote configured. - """ - if cwd is None: - cwd = os.getcwd() - try: - result = subprocess.run( - ["git", "config", "--get", "remote.origin.url"], - capture_output=True, - text=True, - check=True, - cwd=cwd, - ) - url = result.stdout.strip() - if not url: - return "" - digest = hashlib.sha256(url.encode()).hexdigest()[:16] - return f"remote:{digest}" - except (subprocess.CalledProcessError, OSError): - return "" - - -def _remote_url_to_slug(url: str) -> str: - """Convert a git remote URL to a deterministic slug. - - Handles: - - HTTPS: https://github.com/owner/repo.git - - SSH: git@github.com:owner/repo.git - - SSH: git@github.com-alias:owner/repo.git (custom host aliases) - - ssh://: ssh://git@github.com/owner/repo.git - - git://: git://github.com/owner/repo.git - """ - slug = url.strip() - # Strip .git suffix - if slug.endswith(".git"): - slug = slug[:-4] - # Strip protocol prefixes - for prefix in ("https://", "http://", "ssh://", "git://"): - if slug.startswith(prefix): - slug = slug[len(prefix):] - break - else: - # Handle git@ style (no protocol prefix matched) - slug = re.sub(r"^git@", "", slug) - # Replace the first colon (SSH host:path separator) with / - slug = slug.replace(":", "/", 1) - # Split on / and take last two components (owner, repo) - parts = [p for p in slug.split("/") if p] - if len(parts) >= 2: - owner, repo = parts[-2], parts[-1] - slug = f"{owner}-{repo}" - elif parts: - slug = parts[-1] - else: - return "" - # Replace any remaining / and : with - - slug = slug.replace("/", "-").replace(":", "-") - return slug diff --git a/integrations/mem0-plugin/scripts/_project.sh b/integrations/mem0-plugin/scripts/_project.sh deleted file mode 100755 index c4419ac3b..000000000 --- a/integrations/mem0-plugin/scripts/_project.sh +++ /dev/null @@ -1,77 +0,0 @@ -# Source this file. Sets MEM0_PROJECT_ID and MEM0_BRANCH. -# -# Resolution priority (project_id): -# 1. MEM0_PROJECT_ID env var (explicit override) -# 2. ~/.mem0/project_map.json lookup by $PWD (requires jq) -# 3. Git remote slug: strip protocol/prefix, strip .git, replace / and : with - -# e.g. git@github.com:mem0ai/mem0.git -> mem0ai-mem0 -# 4. Fallback: basename of $PWD -# -# Branch resolution: -# git branch --show-current, fallback "unknown" - -_mem0_resolve_project_id() { - # 1. Explicit override - if [ -n "${MEM0_PROJECT_ID:-}" ]; then - printf '%s' "$MEM0_PROJECT_ID" - return - fi - - # 2. project_map.json lookup by $PWD - _mem0_map="$HOME/.mem0/project_map.json" - if [ -f "$_mem0_map" ] && command -v jq >/dev/null 2>&1; then - _mem0_mapped=$(jq -r --arg cwd "$PWD" '.[$cwd] // empty' "$_mem0_map" 2>/dev/null) - if [ -n "$_mem0_mapped" ]; then - printf '%s' "$_mem0_mapped" - return - fi - fi - - # 3. Git remote slug - _mem0_remote_url=$(git remote get-url origin 2>/dev/null) - if [ -n "$_mem0_remote_url" ]; then - _mem0_slug="$_mem0_remote_url" - # Strip .git suffix - _mem0_slug="${_mem0_slug%.git}" - # Strip protocol prefixes - _mem0_slug="${_mem0_slug#https://}" - _mem0_slug="${_mem0_slug#http://}" - _mem0_slug="${_mem0_slug#ssh://}" - _mem0_slug="${_mem0_slug#git://}" - _mem0_slug="${_mem0_slug#git@}" - # Replace first colon (SSH host:path separator) with / - # shellcheck disable=SC2039 - _mem0_slug="${_mem0_slug/://}" - # Keep only the last two path components (owner/repo) - _mem0_owner=$(printf '%s' "$_mem0_slug" | awk -F'/' '{print $(NF-1)}') - _mem0_repo=$(printf '%s' "$_mem0_slug" | awk -F'/' '{print $NF}') - _mem0_slug="${_mem0_owner}-${_mem0_repo}" - # Replace any remaining / and : with - - # shellcheck disable=SC2039 - _mem0_slug="${_mem0_slug//\//-}" - # shellcheck disable=SC2039 - _mem0_slug="${_mem0_slug//:/-}" - if [ -n "$_mem0_slug" ]; then - printf '%s' "$_mem0_slug" - _MEM0_PERSIST_CWD="$PWD" _MEM0_PERSIST_SLUG="$_mem0_slug" python3 -c " -import os, sys -sys.path.insert(0, '$(dirname "${BASH_SOURCE[0]:-$0}")') -from _project import save_project_mapping -save_project_mapping(os.environ['_MEM0_PERSIST_CWD'], os.environ['_MEM0_PERSIST_SLUG']) -" 2>/dev/null || true - return - fi - fi - - # 4. Fallback: basename of $PWD - printf '%s' "$(basename "$PWD")" -} - -_mem0_resolve_branch() { - git branch --show-current 2>/dev/null || printf 'unknown' -} - -MEM0_PROJECT_ID="$(_mem0_resolve_project_id)" -MEM0_BRANCH="$(_mem0_resolve_branch)" -export MEM0_PROJECT_ID -export MEM0_BRANCH diff --git a/integrations/mem0-plugin/scripts/_search.py b/integrations/mem0-plugin/scripts/_search.py deleted file mode 100644 index 95db89ce9..000000000 --- a/integrations/mem0-plugin/scripts/_search.py +++ /dev/null @@ -1,105 +0,0 @@ -"""Shared mem0 search API helper. - -Wraps POST /v3/memories/search/ into a single function call. -All pre-fetch hooks use this instead of duplicating urllib boilerplate. -""" - -from __future__ import annotations - -import json -import os -import sys -import urllib.request - -SEARCH_URL = "https://api.mem0.ai/v3/memories/search/" -SEARCH_TIMEOUT = 5 - - -def should_rerank() -> bool: - """Whether auto-injection searches should request Platform reranking. - - The REST search endpoint does not rerank when ``rerank`` is omitted, so - auto-injected context is ordered by raw vector similarity and the single - most relevant memory can fall outside the injected top_k window. We default - reranking ON for the hook-driven injection path (the extra ~150-200ms is - well within the hook's curl budget) and let users opt out via MEM0_RERANK. - - MEM0_RERANK is read case-insensitively; ``0``, ``false``, ``no``, and - ``off`` disable reranking. Anything else (including unset) enables it. - """ - raw = os.environ.get("MEM0_RERANK") - if raw is None: - return True - return raw.strip().lower() not in ("0", "false", "no", "off", "") - - -def _do_search(api_key: str, payload: dict) -> list[dict]: - body = json.dumps(payload).encode() - req = urllib.request.Request( - SEARCH_URL, - data=body, - headers={"Authorization": f"Token {api_key}", "Content-Type": "application/json"}, - method="POST", - ) - with urllib.request.urlopen(req, timeout=SEARCH_TIMEOUT) as r: - data = json.loads(r.read()) - return data if isinstance(data, list) else data.get("results", []) - - -def search_memories( - api_key: str, - user_id: str, - project_id: str, - query: str, - metadata_type: str | None = None, - metadata_filters: dict | None = None, - top_k: int = 3, - min_score: float = 0.0, - rerank: bool = False, - threshold: float = 0.3, - global_search: bool = False, -) -> list[dict]: - if not api_key: - return [] - - if global_search: - filters: dict = {"OR": [{"user_id": "*"}]} - else: - base_clauses: list[dict] = [{"user_id": user_id}, {"app_id": project_id}] - if metadata_type: - base_clauses.append({"metadata": {"type": metadata_type}}) - if metadata_filters: - for key, value in metadata_filters.items(): - base_clauses.append({"metadata": {key: value}}) - filters = {"AND": base_clauses} - - base_payload: dict = {"query": query, "top_k": top_k, "threshold": threshold} - if rerank: - base_payload["rerank"] = True - - try: - payload = {**base_payload, "filters": filters} - results = _do_search(api_key, payload)[:top_k] - - if min_score > 0: - results = [m for m in results if m.get("score", 0) >= min_score] - return results - except Exception as e: - print(f"[mem0] search request failed: {e}", file=sys.stderr) - return [] - - -def format_results_for_context( - memories: list[dict], - heading: str = "Relevant memories", -) -> str: - if not memories: - return "" - lines = [f"### {heading}", ""] - for m in memories: - mid = m.get("id", "?")[:8] - text = m.get("memory", "")[:200] - cat = (m.get("metadata") or {}).get("type", "unknown") - lines.append(f"- [{cat}] {text} [mem0:{mid}]") - lines.append("") - return "\n".join(lines) diff --git a/integrations/mem0-plugin/scripts/auto_capture.py b/integrations/mem0-plugin/scripts/auto_capture.py deleted file mode 100755 index b48a928b4..000000000 --- a/integrations/mem0-plugin/scripts/auto_capture.py +++ /dev/null @@ -1,213 +0,0 @@ -#!/usr/bin/env python3 -"""Auto-capture recent conversation exchanges into mem0. - -Runs in the background from UserPromptSubmit hook (every 3rd message). -Reads the last few exchanges from the transcript, sends them to the -mem0 API with infer=True so the platform extracts facts automatically. - -Input: env vars (MEM0_API_KEY, MEM0_RESOLVED_USER_ID, MEM0_PROJECT_ID, etc.) - argv[1] = transcript_path -Output: stderr logs only (exit 0 always — must not block) -""" - -from __future__ import annotations - -import json -import logging -import os -import sys -import urllib.error -import urllib.request - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _identity import resolve_api_key, resolve_user_id -from _instructions import load_instructions -from _project import resolve_branch, resolve_project_id - -log = logging.getLogger("mem0-auto-capture") -log.setLevel(logging.DEBUG) -_handler = logging.StreamHandler(sys.stderr) -_handler.setFormatter(logging.Formatter("[mem0-auto-capture] %(message)s")) -log.addHandler(_handler) - -if os.environ.get("MEM0_DEBUG"): - _log_dir = os.path.expanduser("~/.mem0") - try: - os.makedirs(_log_dir, exist_ok=True) - _fh = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) - _fh.setFormatter(logging.Formatter("[mem0-auto-capture] %(asctime)s %(message)s")) - log.addHandler(_fh) - except OSError: - pass - -API_URL = "https://api.mem0.ai" -TAIL_LINES = 200 -MAX_CONTENT_CHARS = 8000 -MIN_CONTENT_CHARS = 100 - - -def tail_lines(filepath: str, n: int) -> list[str]: - try: - with open(filepath, "rb") as f: - f.seek(0, 2) - file_size = f.tell() - if file_size == 0: - return [] - chunk_size = min(file_size, n * 4096) - f.seek(max(0, file_size - chunk_size)) - data = f.read().decode("utf-8", errors="replace") - return data.splitlines()[-n:] - except OSError: - return [] - - -def extract_recent_exchanges(lines: list[str], max_exchanges: int = 3) -> list[dict]: - """Extract the last N user+assistant message pairs from the transcript JSONL.""" - messages = [] - for line in lines: - line = line.strip() - if not line: - continue - try: - entry = json.loads(line) - except json.JSONDecodeError: - continue - - if entry.get("isCompactSummary"): - continue - - msg = entry.get("message", {}) - role = msg.get("role", "") - if role not in ("user", "assistant"): - continue - - content = msg.get("content", "") - if isinstance(content, list): - parts = [] - for block in content: - if isinstance(block, str): - parts.append(block) - elif isinstance(block, dict) and block.get("type") == "text": - parts.append(block.get("text", "")) - content = "\n".join(parts).strip() - - if not content or len(content) < 20: - continue - - # Skip tool-call-only assistant messages - if role == "assistant" and content.startswith("{"): - continue - - messages.append({"role": role, "content": content[:2000]}) - - # Take last N exchanges (pairs of user+assistant) - if not messages: - return [] - - result = messages[-(max_exchanges * 2):] - return result - - -def store_exchange(api_key: str, messages: list[dict], user_id: str, - project_id: str, branch: str, session_id: str) -> bool: - metadata = { - "type": "auto_capture", - "source": "auto_capture", - "confidence": 0.7, - } - if branch: - metadata["branch"] = branch - if session_id: - metadata["session_id"] = session_id - - body = { - "messages": messages, - "user_id": user_id, - "app_id": project_id, - "metadata": metadata, - "infer": True, - } - # Apply the project's mem0.md extraction policy (custom/agent instructions). - body.update(load_instructions()) - - data = json.dumps(body).encode("utf-8") - req = urllib.request.Request( - f"{API_URL}/v3/memories/add/", - data=data, - headers={ - "Content-Type": "application/json", - "Authorization": f"Token {api_key}", - }, - method="POST", - ) - try: - with urllib.request.urlopen(req, timeout=15) as resp: - if resp.status in (200, 201): - result = json.loads(resp.read()) - log.info("Auto-captured: event_id=%s status=%s", - result.get("event_id", "?"), result.get("status", "?")) - return True - log.warning("API returned status %d", resp.status) - return False - except urllib.error.URLError as e: - log.warning("API call failed: %s", e) - return False - - -def main(): - api_key = resolve_api_key() - if not api_key: - log.debug("MEM0_API_KEY not set, skipping") - return - - if len(sys.argv) < 2: - log.debug("No transcript_path argument") - return - - transcript_path = sys.argv[1] - if not transcript_path or not os.path.isfile(transcript_path): - log.debug("Transcript not found: %s", transcript_path) - return - - user_id = resolve_user_id() - project_id = resolve_project_id() - branch = resolve_branch() - session_id = "" - sid_file = f"/tmp/mem0_session_id_{os.environ.get('USER', 'default')}" - if os.path.isfile(sid_file): - try: - with open(sid_file) as f: - session_id = f.read().strip() - except OSError: - pass - - lines = tail_lines(transcript_path, TAIL_LINES) - if not lines: - log.debug("Transcript empty") - return - - messages = extract_recent_exchanges(lines, max_exchanges=4) - if not messages: - log.debug("No substantial exchanges found") - return - - total_chars = sum(len(m["content"]) for m in messages) - if total_chars < MIN_CONTENT_CHARS: - log.debug("Exchanges too short (%d chars), skipping", total_chars) - return - - log.info("Auto-capturing %d messages (%d chars)", len(messages), total_chars) - if store_exchange(api_key, messages, user_id, project_id, branch, session_id): - try: - import session_stats - session_stats.record_add("auto_capture") - except Exception: - pass - - -if __name__ == "__main__": - try: - main() - except Exception as e: - log.error("Unexpected error: %s", e) - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/auto_import.py b/integrations/mem0-plugin/scripts/auto_import.py deleted file mode 100644 index 004772bbd..000000000 --- a/integrations/mem0-plugin/scripts/auto_import.py +++ /dev/null @@ -1,374 +0,0 @@ -#!/usr/bin/env python3 -"""Auto-import declarative project files into mem0. - -Runs in the background from the SessionStart hook (startup only). -Imports CLAUDE.md, AGENTS.md, .cursorrules, .windsurfrules, mem0.md -into mem0 as project profile memories, skipping unchanged files via -SHA-256 hashing. - -Input: MEM0_CWD env var (optional, defaults to os.getcwd()) -Output: stderr logs only (exit 0 always — must not block) -""" - -from __future__ import annotations - -import hashlib -import json -import logging -import os -import sys -import urllib.error -import urllib.request - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _chunking import filter_and_truncate, split_by_headers -from _identity import resolve_api_key, resolve_user_id -from _project import resolve_branch, resolve_project_id, save_project_mapping - -log = logging.getLogger("mem0-auto-import") -log.setLevel(logging.DEBUG) -_handler = logging.StreamHandler(sys.stderr) -_handler.setFormatter(logging.Formatter("[mem0-auto-import] %(message)s")) -log.addHandler(_handler) - -if os.environ.get("MEM0_DEBUG"): - _log_dir = os.path.expanduser("~/.mem0") - try: - os.makedirs(_log_dir, exist_ok=True) - _file_handler = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) - _file_handler.setFormatter(logging.Formatter("[mem0-auto-import] %(asctime)s %(message)s")) - log.addHandler(_file_handler) - except OSError: - pass - -API_URL = "https://api.mem0.ai" -MAX_FILE_SIZE = 100_000 # skip files over 100 KB -TARGET_FILES = ["CLAUDE.md", "AGENTS.md", ".cursorrules", ".windsurfrules", "mem0.md"] -HASH_STORE = os.path.expanduser("~/.mem0/file_hashes.json") -LOCK_FILE = os.path.expanduser("~/.mem0/auto_import.lock") - - -def _acquire_lock() -> bool: - """Try to acquire a file lock. Returns False if another instance is running.""" - try: - os.makedirs(os.path.dirname(LOCK_FILE), exist_ok=True) - fd = os.open(LOCK_FILE, os.O_CREAT | os.O_EXCL | os.O_WRONLY) - os.write(fd, str(os.getpid()).encode()) - os.close(fd) - return True - except FileExistsError: - try: - mtime = os.path.getmtime(LOCK_FILE) - import time - if time.time() - mtime > 120: - os.unlink(LOCK_FILE) - return _acquire_lock() - except OSError: - pass - return False - - -def _release_lock() -> None: - try: - os.unlink(LOCK_FILE) - except OSError: - pass - - -def _git_root(cwd: str) -> str: - """Return the git repo root, or empty string if not in a repo.""" - import subprocess - try: - result = subprocess.run( - ["git", "rev-parse", "--show-toplevel"], - cwd=cwd, capture_output=True, text=True, timeout=5, - ) - if result.returncode == 0: - return result.stdout.strip() - except (OSError, subprocess.TimeoutExpired): - pass - return "" - - -def sha256_file(path: str) -> str: - """Return the hex SHA-256 digest of a file.""" - h = hashlib.sha256() - with open(path, "rb") as f: - for chunk in iter(lambda: f.read(65536), b""): - h.update(chunk) - return h.hexdigest() - - -def load_hashes() -> dict[str, str]: - """Load the hash store from disk; return empty dict on any error.""" - if not os.path.isfile(HASH_STORE): - return {} - try: - with open(HASH_STORE) as f: - return json.load(f) - except (OSError, json.JSONDecodeError): - return {} - - -def save_hashes(hashes: dict[str, str]) -> None: - """Persist the hash store to disk.""" - mem0_dir = os.path.expanduser("~/.mem0") - os.makedirs(mem0_dir, exist_ok=True) - try: - with open(HASH_STORE, "w") as f: - json.dump(hashes, f, indent=2) - except OSError as e: - log.warning("Could not save hash store: %s", e) - - -def already_imported(api_key: str, user_id: str, project_id: str, filename: str) -> bool: - body = json.dumps({ - "query": filename, - "filters": { - "AND": [ - {"user_id": user_id}, - {"app_id": project_id}, - {"metadata": {"source": "auto-import"}}, - ] - }, - "top_k": 10, - "threshold": 0.0, - }).encode() - req = urllib.request.Request( - f"{API_URL}/v3/memories/search/", - data=body, - headers={"Content-Type": "application/json", "Authorization": f"Token {api_key}"}, - method="POST", - ) - try: - with urllib.request.urlopen(req, timeout=5) as r: - data = json.loads(r.read()) - results = data if isinstance(data, list) else data.get("results", []) - for result in results: - meta = result.get("metadata", {}) if isinstance(result, dict) else {} - file_field = meta.get("file", "") - if file_field == filename or file_field.startswith(f"{filename}["): - return True - return False - except Exception: - return False - - -def _delete_stale_chunks(api_key: str, user_id: str, project_id: str, filename: str) -> int: - """Find and delete existing chunks for a file before re-import. Returns count deleted.""" - body = json.dumps({ - "query": filename, - "filters": { - "AND": [ - {"user_id": user_id}, - {"app_id": project_id}, - {"metadata": {"source": "auto-import"}}, - ] - }, - "top_k": 20, - "threshold": 0.0, - }).encode() - req = urllib.request.Request( - f"{API_URL}/v3/memories/search/", - data=body, - headers={"Content-Type": "application/json", "Authorization": f"Token {api_key}"}, - method="POST", - ) - ids_to_delete = [] - try: - with urllib.request.urlopen(req, timeout=10) as r: - data = json.loads(r.read()) - results = data if isinstance(data, list) else data.get("results", []) - for result in results: - if not isinstance(result, dict): - continue - meta = result.get("metadata", {}) - file_field = meta.get("file", "") - if file_field == filename or file_field.startswith(f"{filename}["): - mid = result.get("id") - if mid: - ids_to_delete.append(mid) - except Exception as e: - log.warning("Failed to search for stale chunks of %s: %s", filename, e) - return 0 - - deleted = 0 - for mid in ids_to_delete: - try: - del_req = urllib.request.Request( - f"{API_URL}/v1/memories/{mid}/", - headers={"Authorization": f"Token {api_key}"}, - method="DELETE", - ) - with urllib.request.urlopen(del_req, timeout=10): - deleted += 1 - except Exception as e: - log.warning("Failed to delete stale chunk %s: %s", mid, e) - - if deleted: - log.info("Deleted %d stale chunk(s) for %s before re-import", deleted, filename) - return deleted - - -def post_memory(api_key: str, content: str, user_id: str, filename: str, project_id: str, branch: str = "") -> bool: - """POST a project profile memory to the Mem0 REST API.""" - metadata = { - "type": "project_profile", - "file": filename, - "source": "auto-import", - } - if branch: - metadata["branch"] = branch - body = { - "messages": [ - { - "role": "user", - "content": f"## Project Profile: {filename}\n\nProject: {project_id}\n\n{content}", - } - ], - "user_id": user_id, - "app_id": project_id, - "metadata": metadata, - "infer": False, - } - - data = json.dumps(body).encode("utf-8") - req = urllib.request.Request( - f"{API_URL}/v3/memories/add/", - data=data, - headers={ - "Content-Type": "application/json", - "Authorization": f"Token {api_key}", - }, - method="POST", - ) - - try: - with urllib.request.urlopen(req, timeout=15) as resp: - if resp.status in (200, 201): - log.info("Imported %s (project=%s)", filename, project_id) - return True - log.warning("API returned status %d for %s", resp.status, filename) - return False - except urllib.error.URLError as e: - log.warning("API call failed for %s: %s", filename, e) - return False - - -def main() -> None: - api_key = resolve_api_key() - if not api_key: - log.debug("MEM0_API_KEY not set, skipping auto-import") - return - - cwd = os.environ.get("MEM0_CWD", "").strip() or os.getcwd() - user_id = resolve_user_id() - project_id = resolve_project_id(cwd) - branch = resolve_branch(cwd) - - save_project_mapping(cwd, project_id) - - git_root = _git_root(cwd) - search_dirs = [cwd] - if git_root and os.path.realpath(git_root) != os.path.realpath(cwd): - search_dirs.append(git_root) - - log.debug("Auto-import started: cwd=%s git_root=%s project=%s user=%s branch=%s", cwd, git_root or "(none)", project_id, user_id, branch) - - hashes = load_hashes() - updated = False - seen_content_hashes: set[str] = set() - - for filename in TARGET_FILES: - filepath = "" - for search_dir in search_dirs: - candidate = os.path.join(search_dir, filename) - if os.path.isfile(candidate): - filepath = candidate - break - - if not filepath: - log.debug("Not found, skipping: %s", filename) - continue - - filepath = os.path.realpath(filepath) - - try: - file_size = os.path.getsize(filepath) - except OSError: - log.debug("Cannot stat %s, skipping", filename) - continue - - if file_size > MAX_FILE_SIZE: - log.debug("Skipping %s: size %d exceeds %d bytes", filename, file_size, MAX_FILE_SIZE) - continue - - try: - current_hash = sha256_file(filepath) - except OSError as e: - log.debug("Cannot hash %s: %s", filename, e) - continue - - if current_hash in seen_content_hashes: - log.debug("Duplicate content, skipping: %s (same as earlier file)", filename) - continue - seen_content_hashes.add(current_hash) - - hash_key = f"{project_id}:{branch}:{filename}" if branch else f"{project_id}:{filename}" - if hashes.get(hash_key) == current_hash: - if already_imported(api_key, user_id, project_id, filename): - log.debug("Unchanged and still in mem0, skipping: %s", filename) - continue - log.info("Hash matches but memories missing server-side, re-importing: %s", filename) - - elif already_imported(api_key, user_id, project_id, filename): - log.debug("Already in mem0, updating hash store: %s", filename) - hashes[hash_key] = current_hash - updated = True - continue - - try: - with open(filepath, encoding="utf-8", errors="replace") as f: - content = f.read() - except OSError as e: - log.debug("Cannot read %s: %s", filename, e) - continue - - _delete_stale_chunks(api_key, user_id, project_id, filename) - - is_markdown = filename.endswith(".md") - if is_markdown: - chunks = filter_and_truncate(split_by_headers(content)) - else: - chunks = filter_and_truncate([content]) - - if not chunks: - chunks = [content[:10000]] - - success = True - for i, chunk in enumerate(chunks): - chunk_name = f"{filename}[{i+1}/{len(chunks)}]" if len(chunks) > 1 else filename - if not post_memory(api_key, chunk, user_id, chunk_name, project_id, branch): - success = False - - if success: - hashes[hash_key] = current_hash - updated = True - - if updated: - save_hashes(hashes) - else: - log.debug("No files imported this run") - - -if __name__ == "__main__": - if not _acquire_lock(): - log.debug("Another auto_import instance is running — skipping") - sys.exit(0) - try: - main() - except Exception as e: - log.error("Unexpected error: %s", e) - finally: - _release_lock() - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/auto_setup_categories.py b/integrations/mem0-plugin/scripts/auto_setup_categories.py deleted file mode 100644 index a94ed00a5..000000000 --- a/integrations/mem0-plugin/scripts/auto_setup_categories.py +++ /dev/null @@ -1,236 +0,0 @@ -#!/usr/bin/env python3 -"""Auto-configure mem0's coding-category taxonomy in the background. - -Runs from the SessionStart hook (startup only), exactly like auto_import.py. -mem0 auto-tags every memory with `categories`; by default that list is -consumer-oriented (food, hobbies, ...), which is useless for code. This script -replaces it with the coding-focused taxonomy defined in -``setup_coding_categories.CODING_CATEGORIES`` so search and retrieval are tuned -for development work — without the user ever being asked during onboarding. - -Design (mirrors auto_import.py): - - Resolve the API key; do nothing if it is absent. - - Gate on a state file (``~/.mem0/categories_setup.json``) keyed by a hash of - the API key -> a hash of the taxonomy. Categories are scoped to the mem0 - *project* tied to the API key (NOT to the local repo), so this only needs to - run once per account, and re-runs only if the taxonomy itself changes. - - Hold a lock file so concurrent sessions don't race. - - Reuse the proven SDK path (``client.project.update``) via the plugin venv. - - Always exit 0; log to stderr only. Must never block a session. - -Run with no arguments (background) or in the foreground for onboarding to print -a parseable status line. - -Requires MEM0_API_KEY (or CLAUDE_PLUGIN_OPTION_API_KEY) and the mem0ai SDK, -which ensure_deps.sh installs into ${CLAUDE_PLUGIN_DATA}/venv on session start. -""" - -from __future__ import annotations - -import hashlib -import json -import logging -import os -import sys -import time - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -# Importing setup_coding_categories also injects the plugin venv's site-packages -# onto sys.path (its module-level bootstrap), so ``from mem0 import MemoryClient`` -# works even when this script is run with the system python3. -from _identity import resolve_api_key # noqa: E402 -from setup_coding_categories import CODING_CATEGORIES, _categories_match # noqa: E402 - -log = logging.getLogger("mem0-auto-categories") -log.setLevel(logging.DEBUG) -_handler = logging.StreamHandler(sys.stderr) -_handler.setFormatter(logging.Formatter("[mem0-auto-categories] %(message)s")) -log.addHandler(_handler) - -if os.environ.get("MEM0_DEBUG"): - _log_dir = os.path.expanduser("~/.mem0") - try: - os.makedirs(_log_dir, exist_ok=True) - _file_handler = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) - _file_handler.setFormatter(logging.Formatter("[mem0-auto-categories] %(asctime)s %(message)s")) - log.addHandler(_file_handler) - except OSError: - pass - -STATE_FILE = os.path.expanduser("~/.mem0/categories_setup.json") -LOCK_FILE = os.path.expanduser("~/.mem0/categories_setup.lock") - - -# --------------------------------------------------------------------------- # -# Fingerprints # -# --------------------------------------------------------------------------- # -def categories_fingerprint(categories: list = CODING_CATEGORIES) -> str: - """Stable, order-independent 16-hex digest of the category taxonomy. - - Reordering the categories yields the same fingerprint; adding, removing, or - editing a category changes it (so the taxonomy re-applies on upgrade). - """ - pairs = sorted( - (str(key), str(value)) - for entry in categories - if isinstance(entry, dict) - for key, value in entry.items() - ) - payload = json.dumps(pairs, sort_keys=True, ensure_ascii=False) - return hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16] - - -def apikey_fingerprint(api_key: str) -> str: - """Opaque 16-hex digest of the API key. Never stores the key itself.""" - return hashlib.sha256(api_key.encode("utf-8")).hexdigest()[:16] - - -# --------------------------------------------------------------------------- # -# State file # -# --------------------------------------------------------------------------- # -def load_state(path: str = STATE_FILE) -> dict: - """Load the apikey-fingerprint -> categories-fingerprint map; {} on any error.""" - if not os.path.isfile(path): - return {} - try: - with open(path) as f: - data = json.load(f) - return data if isinstance(data, dict) else {} - except (OSError, json.JSONDecodeError): - return {} - - -def save_state(state: dict, path: str = STATE_FILE) -> None: - """Persist the state map, creating the parent directory if needed.""" - parent = os.path.dirname(path) - if parent: - try: - os.makedirs(parent, exist_ok=True) - except OSError as e: - log.warning("Could not create state dir: %s", e) - return - try: - with open(path, "w") as f: - json.dump(state, f, indent=2) - except OSError as e: - log.warning("Could not save categories state: %s", e) - - -def is_applied(state: dict, key_fp: str, cat_fp: str) -> bool: - """True only when this API key has already had this exact taxonomy applied.""" - return state.get(key_fp) == cat_fp - - -# --------------------------------------------------------------------------- # -# SDK interaction (client injected for testability) # -# --------------------------------------------------------------------------- # -def make_client(): - """Construct a MemoryClient. Imported lazily so this module loads without the SDK.""" - from mem0 import MemoryClient - - return MemoryClient() - - -def fetch_current_categories(client) -> list | None: - """Return the project's current custom_categories, or None if unavailable.""" - current = client.project.get(fields=["custom_categories"]) - if isinstance(current, dict): - return current.get("custom_categories") - return None - - -def apply_categories(client, proposed: list = CODING_CATEGORIES) -> str: - """Install the coding taxonomy if it isn't already in place. - - Returns "already-configured" when the project already matches (no write), or - "applied" after a successful ``project.update``. Raises on API failure. - """ - current = fetch_current_categories(client) - if _categories_match(current, proposed): - return "already-configured" - client.project.update(custom_categories=proposed) - return "applied" - - -# --------------------------------------------------------------------------- # -# Lock (mirrors auto_import.py) # -# --------------------------------------------------------------------------- # -def _acquire_lock() -> bool: - """Try to acquire a file lock. Returns False if another instance is running.""" - try: - os.makedirs(os.path.dirname(LOCK_FILE), exist_ok=True) - fd = os.open(LOCK_FILE, os.O_CREAT | os.O_EXCL | os.O_WRONLY) - os.write(fd, str(os.getpid()).encode()) - os.close(fd) - return True - except FileExistsError: - try: - if time.time() - os.path.getmtime(LOCK_FILE) > 120: - os.unlink(LOCK_FILE) - return _acquire_lock() - except OSError: - pass - return False - - -def _release_lock() -> None: - try: - os.unlink(LOCK_FILE) - except OSError: - pass - - -# --------------------------------------------------------------------------- # -# Entry point # -# --------------------------------------------------------------------------- # -def main() -> None: - api_key = resolve_api_key() - if not api_key: - log.debug("MEM0_API_KEY not set, skipping coding-categories setup") - return - - key_fp = apikey_fingerprint(api_key) - cat_fp = categories_fingerprint() - - state = load_state() - if is_applied(state, key_fp, cat_fp): - log.debug("Coding categories already configured for this account (cached); skipping") - return - - os.environ["MEM0_API_KEY"] = api_key - - try: - client = make_client() - except ImportError: - log.debug("mem0ai SDK not ready yet (venv installing?); will retry next session") - return - except Exception as e: - log.warning("Could not initialise MemoryClient: %s", e) - return - - try: - result = apply_categories(client) - except Exception as e: - log.warning("Could not configure coding categories: %s", e) - return - - state[key_fp] = cat_fp - save_state(state) - - if result == "applied": - log.info("Applied %d coding categories", len(CODING_CATEGORIES)) - else: - log.info("Coding categories already configured") - - -if __name__ == "__main__": - if not _acquire_lock(): - log.debug("Another auto_setup_categories instance is running — skipping") - sys.exit(0) - try: - main() - except Exception as e: # never block a session - log.error("Unexpected error: %s", e) - finally: - _release_lock() - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/block_memory_write.sh b/integrations/mem0-plugin/scripts/block_memory_write.sh deleted file mode 100755 index 1b3802663..000000000 --- a/integrations/mem0-plugin/scripts/block_memory_write.sh +++ /dev/null @@ -1,36 +0,0 @@ -#!/usr/bin/env bash -# Hook: PreToolUse (matcher: Write|Edit) -# -# Blocks writes to MEMORY.md and auto-memory files, redirecting Claude -# to use the mem0 MCP add_memory tool instead. -# -# Input: JSON on stdin with tool_name, tool_input -# Output: stderr message (exit 2 = block) -# -# Exit codes: -# 0 = allow the tool call -# 2 = block the tool call (stderr is shown to Claude as feedback) - -set -euo pipefail - -if [ -n "${MEM0_DEBUG:-}" ]; then - mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" -fi - -INPUT=$(cat) - -FILE_PATH=$(echo "$INPUT" | jq -r '.tool_input.file_path // .tool_input.path // ""' 2>/dev/null || echo "") - -if [ -z "$FILE_PATH" ]; then - exit 0 -fi - -case "$FILE_PATH" in - */.claude/*/MEMORY.md|*/.claude/memory/*) - echo "BLOCKED: Do not write to $FILE_PATH. Use the mem0 MCP \`add_memory\` tool instead to persist memories. This project uses mem0 for all memory storage." >&2 - exit 2 - ;; - *) - exit 0 - ;; -esac diff --git a/integrations/mem0-plugin/scripts/block_memory_write_cursor.sh b/integrations/mem0-plugin/scripts/block_memory_write_cursor.sh deleted file mode 100755 index 21b2cf0f8..000000000 --- a/integrations/mem0-plugin/scripts/block_memory_write_cursor.sh +++ /dev/null @@ -1,33 +0,0 @@ -#!/usr/bin/env bash -# Hook: preToolUse (Cursor) — blocks writes to MEMORY.md -# -# Cursor variant of block_memory_write.sh. Returns JSON: -# {"permission":"deny","agent_message":"..."} on block, -# {"permission":"allow"} on pass. - -set -uo pipefail - -if [ -n "${MEM0_DEBUG:-}" ]; then - mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" -fi - -INPUT=$(cat) - -FILE_PATH=$(echo "$INPUT" | jq -r '.tool_input.file_path // .tool_input.path // ""' 2>/dev/null || echo "") - -if [ -z "$FILE_PATH" ]; then - jq -cn '{permission:"allow"}' - exit 0 -fi - -case "$FILE_PATH" in - */MEMORY.md|*/.claude/memory/*|*/.cursor/memory/*) - jq -cn --arg msg "Do not write to $FILE_PATH. Use the mem0 MCP add_memory tool instead to persist memories." \ - '{permission:"deny", agent_message:$msg}' - exit 0 - ;; - *) - jq -cn '{permission:"allow"}' - exit 0 - ;; -esac diff --git a/integrations/mem0-plugin/scripts/capture_compact_summary.py b/integrations/mem0-plugin/scripts/capture_compact_summary.py deleted file mode 100644 index d2f7de0ba..000000000 --- a/integrations/mem0-plugin/scripts/capture_compact_summary.py +++ /dev/null @@ -1,203 +0,0 @@ -#!/usr/bin/env python3 -"""Capture the post-compaction summary into mem0. - -PreCompact hooks fire BEFORE the summary is generated, so they can't -store the actual compact-summary text. This script runs at -SessionStart with source=compact, reads the transcript, finds the -most recent entry flagged isCompactSummary=true, and stores it as a -memory tagged metadata.type=compact_summary. - -Input: JSON on stdin with transcript_path, session_id, source -Output: stderr logs only (exit 0 always -- must not block) - -Spawned in the background by on_session_start.sh; the user-facing -bootstrap text continues without waiting on the network. -""" - -from __future__ import annotations - -import json -import logging -import os -import sys -import urllib.error -import urllib.request -from datetime import date, timedelta - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _identity import resolve_api_key, resolve_user_id -from _instructions import load_instructions -from _project import resolve_branch, resolve_project_id - -log = logging.getLogger("mem0-compact-summary") -log.setLevel(logging.DEBUG) -_handler = logging.StreamHandler(sys.stderr) -_handler.setFormatter(logging.Formatter("[mem0-compact-summary] %(message)s")) -log.addHandler(_handler) - -if os.environ.get("MEM0_DEBUG"): - _log_dir = os.path.expanduser("~/.mem0") - try: - os.makedirs(_log_dir, exist_ok=True) - _file_handler = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) - _file_handler.setFormatter(logging.Formatter("[mem0-compact-summary] %(asctime)s %(message)s")) - log.addHandler(_file_handler) - except OSError: - pass - -API_URL = "https://api.mem0.ai" -MAX_TAIL_LINES = 2000 -MAX_SUMMARY_CHARS = 50000 -# Compact summaries describe a single session's state -- stale after a quarter. -COMPACT_SUMMARY_EXPIRY_DAYS = 90 - - -def tail_lines(filepath: str, n: int) -> list[str]: - try: - with open(filepath, "rb") as f: - f.seek(0, 2) - file_size = f.tell() - if file_size == 0: - return [] - chunk_size = min(file_size, n * 4096) - f.seek(max(0, file_size - chunk_size)) - data = f.read().decode("utf-8", errors="replace") - return data.splitlines()[-n:] - except OSError: - return [] - - -def find_compact_summary(lines: list[str]) -> str: - """Walk transcript backwards, return text content of the most recent - entry flagged isCompactSummary=true. Empty string if none found.""" - for line in reversed(lines): - line = line.strip() - if not line: - continue - try: - entry = json.loads(line) - except json.JSONDecodeError: - continue - if not entry.get("isCompactSummary"): - continue - - message = entry.get("message", {}) - content = message.get("content", []) - if isinstance(content, str): - return content[:MAX_SUMMARY_CHARS] - if isinstance(content, list): - parts = [] - for block in content: - if isinstance(block, str): - parts.append(block) - elif isinstance(block, dict) and block.get("type") == "text": - parts.append(block.get("text", "")) - return "\n".join(parts).strip()[:MAX_SUMMARY_CHARS] - return "" - - -def store_summary(api_key: str, summary: str, user_id: str, session_id: str, project_id: str = "", branch: str = "", cwd: str | None = None) -> bool: - expires = (date.today() + timedelta(days=COMPACT_SUMMARY_EXPIRY_DAYS)).isoformat() - metadata = { - "type": "compact_summary", - "source": "session-start-compact", - "session_id": session_id, - } - if branch: - metadata["branch"] = branch - # The compact summary is model-authored prose, in the first person and with no - # framing to mark it as such. Under role="user" mem0 reads "I recommend X" as - # the human saying it and stores "User recommends X". - body = { - "messages": [{"role": "assistant", "content": summary}], - "user_id": user_id, - "app_id": project_id, - "metadata": metadata, - "infer": True, - "expiration_date": expires, - } - # Apply the project's mem0.md extraction policy (custom/agent instructions). - body.update(load_instructions(cwd)) - - data = json.dumps(body).encode("utf-8") - req = urllib.request.Request( - f"{API_URL}/v3/memories/add/", - data=data, - headers={ - "Content-Type": "application/json", - "Authorization": f"Token {api_key}", - }, - method="POST", - ) - try: - with urllib.request.urlopen(req, timeout=15) as resp: - if resp.status in (200, 201): - log.info("Compact summary stored") - return True - log.warning("API returned status %d", resp.status) - return False - except urllib.error.URLError as e: - log.warning("API call failed: %s", e) - return False - - -def main(): - api_key = resolve_api_key() - if not api_key: - log.debug("MEM0_API_KEY not set, skipping capture") - return - - try: - hook_input = json.loads(sys.stdin.read()) - except (json.JSONDecodeError, OSError): - log.debug("No valid JSON on stdin") - return - - transcript_path = hook_input.get("transcript_path", "") - if not transcript_path: - log.debug("No transcript_path provided") - return - - session_id = hook_input.get("session_id", "") - cwd = hook_input.get("cwd") or None - user_id = resolve_user_id() - project_id = resolve_project_id(cwd) - branch = resolve_branch(cwd) - - lines = tail_lines(transcript_path, MAX_TAIL_LINES) - if not lines: - log.debug("Transcript empty or unreadable: %s", transcript_path) - return - - summary = find_compact_summary(lines) - if not summary: - log.debug("No isCompactSummary entry found") - return - - if len(summary.strip()) < 100: - log.debug("Compact summary too short (%d chars) — skipping", len(summary.strip())) - return - - marker_dir = os.path.expanduser("~/.mem0") - marker_file = os.path.join(marker_dir, f"compact_captured_{session_id}") - if session_id and os.path.isfile(marker_file): - log.info("Compact summary already captured for session %s — skipping", session_id) - return - - log.info("Capturing compact summary (%d chars)", len(summary)) - if store_summary(api_key, summary, user_id, session_id, project_id, branch, cwd): - if session_id: - try: - os.makedirs(marker_dir, exist_ok=True) - with open(marker_file, "w") as f: - f.write("") - except OSError: - pass - - -if __name__ == "__main__": - try: - main() - except Exception as e: - log.error("Unexpected error: %s", e) - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/capture_session_summary.py b/integrations/mem0-plugin/scripts/capture_session_summary.py deleted file mode 100644 index 6c3fa92fa..000000000 --- a/integrations/mem0-plugin/scripts/capture_session_summary.py +++ /dev/null @@ -1,272 +0,0 @@ -#!/usr/bin/env python3 -"""Capture a structured session summary on Stop hook. - -Runs on every Stop (end of each assistant turn). Each invocation reads -the transcript JSONL, extracts the latest assistant message and all -files touched so far, then stores via mem0 API with infer=True. Uses -run_id=session_id to scope infer dedup to the session, so the final -stored summary reflects the most recent turn — not just the first. - -Input: JSON on stdin with transcript_path, session_id, cwd, agent_id -Output: stderr logs only (exit 0 always — must not block) -""" - -from __future__ import annotations - -import json -import logging -import os -import re -import sys -import urllib.error -import urllib.request -from datetime import date, timedelta - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _identity import resolve_api_key, resolve_user_id -from _instructions import load_instructions -from _project import resolve_branch, resolve_project_id - -log = logging.getLogger("mem0-session-summary") -log.setLevel(logging.DEBUG) -_handler = logging.StreamHandler(sys.stderr) -_handler.setFormatter(logging.Formatter("[mem0-session-summary] %(message)s")) -log.addHandler(_handler) - -if os.environ.get("MEM0_DEBUG"): - _log_dir = os.path.expanduser("~/.mem0") - try: - os.makedirs(_log_dir, exist_ok=True) - _fh = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) - _fh.setFormatter(logging.Formatter("[mem0-session-summary] %(asctime)s %(message)s")) - log.addHandler(_fh) - except OSError: - pass - -API_URL = "https://api.mem0.ai" -MAX_TAIL_LINES = 3000 -MAX_SUMMARY_CHARS = 50000 -SUMMARY_EXPIRY_DAYS = 90 - -SYSTEM_TAG_RE = re.compile( - r"<(?:system-reminder|private|claude-mem-context|persisted-output|system_instruction)>" - r".*?" - r"", - re.DOTALL, -) - - -def tail_lines(filepath: str, n: int) -> list[str]: - try: - with open(filepath, "rb") as f: - f.seek(0, 2) - file_size = f.tell() - if file_size == 0: - return [] - chunk_size = min(file_size, n * 4096) - f.seek(max(0, file_size - chunk_size)) - data = f.read().decode("utf-8", errors="replace") - return data.splitlines()[-n:] - except OSError: - return [] - - -def extract_last_assistant_message(lines: list[str]) -> str: - """Walk transcript backwards, return text content of the last assistant message.""" - for line in reversed(lines): - line = line.strip() - if not line: - continue - if '"type":"assistant"' not in line and '"type": "assistant"' not in line: - continue - try: - entry = json.loads(line) - except json.JSONDecodeError: - continue - if entry.get("type") != "assistant": - continue - message = entry.get("message", {}) - content = message.get("content", []) - if isinstance(content, str): - return content - if isinstance(content, list): - parts = [] - for block in content: - if isinstance(block, str): - parts.append(block) - elif isinstance(block, dict) and block.get("type") == "text": - parts.append(block.get("text", "")) - return "\n".join(parts).strip() - return "" - - -def extract_files_touched(lines: list[str]) -> list[str]: - """Extract unique file paths from tool_use content blocks in transcript.""" - files = set() - file_ext_re = re.compile( - r"[a-zA-Z0-9_./-]+\.(?:py|ts|tsx|js|jsx|rs|go|rb|java|sh|yaml|yml|json|toml|md|sql|css|html)" - ) - for line in lines: - line = line.strip() - if not line: - continue - if '"tool_use"' not in line and '"file_path"' not in line: - continue - try: - entry = json.loads(line) - except json.JSONDecodeError: - continue - content = entry.get("message", {}).get("content", []) - if not isinstance(content, list): - continue - for block in content: - if not isinstance(block, dict) or block.get("type") != "tool_use": - continue - inp = block.get("input", {}) - if not isinstance(inp, dict): - continue - fp = inp.get("file_path", "") - if fp: - files.add(fp) - command = inp.get("command", "") - if command: - for match in file_ext_re.findall(command): - files.add(match) - return sorted(files)[:20] - - -def strip_tags(text: str) -> str: - return SYSTEM_TAG_RE.sub("", text).strip() - - -def build_summary_prompt(assistant_msg: str, files: list[str]) -> str: - """Build a structured prompt that helps mem0's AI extract a good summary.""" - files_section = "" - if files: - file_list = ", ".join(files[:10]) - files_section = f"\n\nFiles touched during this session: {file_list}" - - return ( - f"Session summary — store the following as a structured session summary.\n\n" - f"What the assistant accomplished in this session:\n" - f"{assistant_msg[:MAX_SUMMARY_CHARS]}" - f"{files_section}\n\n" - f"Extract and remember: what was requested, what was investigated, " - f"key decisions made, what was completed, and what needs to happen next." - ) - - -def store_summary( - api_key: str, - summary_prompt: str, - user_id: str, - session_id: str, - project_id: str, - branch: str, - files: list[str], - cwd: str | None = None, -) -> bool: - expires = (date.today() + timedelta(days=SUMMARY_EXPIRY_DAYS)).isoformat() - metadata = { - "type": "session_summary", - "source": "stop-hook", - "session_id": session_id, - } - if branch: - metadata["branch"] = branch - if files: - metadata["files_touched"] = files[:20] - - # summary_prompt wraps the assistant's own last message. Mem0 extracts "facts - # about the user" from each message and role is the only signal telling it who - # spoke, so role="user" here turns Claude's opinions into the human's stated - # preferences ("User prefers dropping Redis..."). - body = { - "messages": [{"role": "assistant", "content": summary_prompt}], - "user_id": user_id, - "app_id": project_id, - "run_id": session_id, - "metadata": metadata, - "infer": True, - "expiration_date": expires, - } - # Apply the project's mem0.md extraction policy (custom/agent instructions). - body.update(load_instructions(cwd)) - - data = json.dumps(body).encode("utf-8") - req = urllib.request.Request( - f"{API_URL}/v3/memories/add/", - data=data, - headers={ - "Content-Type": "application/json", - "Authorization": f"Token {api_key}", - }, - method="POST", - ) - try: - with urllib.request.urlopen(req, timeout=15) as resp: - if resp.status in (200, 201): - log.info("Session summary stored") - return True - log.warning("API returned status %d", resp.status) - return False - except urllib.error.URLError as e: - log.warning("API call failed: %s", e) - return False - - -def main(): - api_key = resolve_api_key() - if not api_key: - log.debug("MEM0_API_KEY not set, skipping") - return - - try: - hook_input = json.loads(sys.stdin.read()) - except (json.JSONDecodeError, OSError): - log.debug("No valid JSON on stdin") - return - - # Guard: skip subagent sessions (only root sessions get summaries) - agent_id = hook_input.get("agent_id", "") - if agent_id: - log.debug("Subagent session (agent_id=%s), skipping", agent_id) - return - - transcript_path = hook_input.get("transcript_path", "") - if not transcript_path: - log.debug("No transcript_path provided") - return - - session_id = hook_input.get("session_id", "") - cwd = hook_input.get("cwd") or None - - user_id = resolve_user_id() - project_id = resolve_project_id(cwd) - branch = resolve_branch(cwd) - - lines = tail_lines(transcript_path, MAX_TAIL_LINES) - if not lines: - log.debug("Transcript empty or unreadable: %s", transcript_path) - return - - assistant_msg = extract_last_assistant_message(lines) - if not assistant_msg or len(assistant_msg.strip()) < 100: - log.debug("Assistant message too short (%d chars) — skipping", len(assistant_msg.strip())) - return - - assistant_msg = strip_tags(assistant_msg) - files = extract_files_touched(lines) - - summary_prompt = build_summary_prompt(assistant_msg, files) - - log.info("Capturing session summary (%d chars, %d files)", len(assistant_msg), len(files)) - store_summary(api_key, summary_prompt, user_id, session_id, project_id, branch, files, cwd) - - -if __name__ == "__main__": - try: - main() - except Exception as e: - log.error("Unexpected error: %s", e) - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/enforce_metadata_defaults.sh b/integrations/mem0-plugin/scripts/enforce_metadata_defaults.sh deleted file mode 100755 index e24b63648..000000000 --- a/integrations/mem0-plugin/scripts/enforce_metadata_defaults.sh +++ /dev/null @@ -1,218 +0,0 @@ -#!/usr/bin/env bash -# PreToolUse hook for mem0 MCP tools. -# Injects identity (user_id, app_id) and metadata defaults when the agent -# omits them. Uses the hookSpecificOutput.updatedInput contract to modify -# tool call parameters before execution. -# -# Handles: -# add_memory — top-level user_id, app_id, metadata defaults -# search_memories — user_id/app_id into filters.AND[] -# get_memories — user_id/app_id into filters.AND[] -# delete_all_memories — top-level user_id, app_id -# -# Hook contract: -# exit 0 = allow. If stdout contains {"hookSpecificOutput": {"updatedInput": ...}}, -# the updatedInput replaces the tool's input parameters. -# exit 2 = block (stderr shown as rejection reason). - -set -uo pipefail - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -source "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true - -INPUT=$(cat) - -TOOL_NAME=$(echo "$INPUT" | jq -r '.tool_name // ""' 2>/dev/null) - -# Determine which handler to use based on tool name -HANDLER="" -case "$TOOL_NAME" in - mcp__mem0__add_memory|mcp__plugin_mem0_mem0__add_memory) - HANDLER="add_memory" ;; - mcp__mem0__search_memories|mcp__plugin_mem0_mem0__search_memories) - HANDLER="search_memories" ;; - mcp__mem0__get_memories|mcp__plugin_mem0_mem0__get_memories) - HANDLER="get_memories" ;; - mcp__mem0__delete_all_memories|mcp__plugin_mem0_mem0__delete_all_memories) - HANDLER="delete_all" ;; - *) exit 0 ;; -esac - -TOOL_INPUT=$(echo "$INPUT" | jq -r '.tool_input // "{}"' 2>/dev/null) - -_PATCH_OUT="/tmp/mem0_enforce_$$" -trap 'rm -f "$_PATCH_OUT"' EXIT -_MEM0_TOOL_INPUT="$TOOL_INPUT" \ -_MEM0_USER_ID="${MEM0_RESOLVED_USER_ID:-}" \ -_MEM0_APP_ID="${MEM0_PROJECT_ID:-}" \ -_MEM0_GLOBAL_SEARCH="${MEM0_GLOBAL_SEARCH:-false}" \ -_MEM0_HANDLER="$HANDLER" \ -python3 <<'PYEOF' > "$_PATCH_OUT" 2>/dev/null || true -import json, os, sys - -raw = os.environ.get("_MEM0_TOOL_INPUT", "{}") -try: - inp = json.loads(raw) -except Exception: - sys.exit(0) - -handler = os.environ.get("_MEM0_HANDLER", "") -resolved_uid = os.environ.get("_MEM0_USER_ID", "") -resolved_aid = os.environ.get("_MEM0_APP_ID", "") -global_search = os.environ.get("_MEM0_GLOBAL_SEARCH", "false") == "true" -changed = False - - -def inject_top_level_identity(inp, uid, aid): - """Inject user_id/app_id as top-level params (for add_memory, delete_all).""" - changed = False - if uid and not inp.get("user_id"): - inp["user_id"] = uid - changed = True - if aid and not inp.get("app_id"): - inp["app_id"] = aid - changed = True - return changed - - -def inject_filter_identity(inp, uid, aid): - """Inject user_id/app_id into filters.AND[] (for search/get_memories).""" - changed = False - if not uid and not aid: - return False - - filters = inp.get("filters") - - if filters is None: - # No filters at all — create from scratch - and_clauses = [] - if uid: - and_clauses.append({"user_id": uid}) - if aid: - and_clauses.append({"app_id": aid}) - inp["filters"] = {"AND": and_clauses} - return True - - if not isinstance(filters, dict): - return False - - # Check if filters already contain user_id/app_id - and_clauses = filters.get("AND") - if and_clauses is None: - # Filters exist but no AND — could be flat like {"user_id": "x"} - has_uid = "user_id" in filters - has_aid = "app_id" in filters - if has_uid and has_aid: - return False - # Convert flat filters to AND format and add missing identity - existing = [] - for k, v in list(filters.items()): - existing.append({k: v}) - if uid and not has_uid: - existing.append({"user_id": uid}) - changed = True - if aid and not has_aid: - existing.append({"app_id": aid}) - changed = True - if changed: - inp["filters"] = {"AND": existing} - return changed - - if not isinstance(and_clauses, list): - return False - - # AND array exists — check for existing user_id/app_id - has_uid = any("user_id" in c for c in and_clauses if isinstance(c, dict)) - has_aid = any("app_id" in c for c in and_clauses if isinstance(c, dict)) - - if uid and not has_uid: - and_clauses.append({"user_id": uid}) - changed = True - if aid and not has_aid: - and_clauses.append({"app_id": aid}) - changed = True - - return changed - - -if handler == "add_memory": - changed = inject_top_level_identity(inp, resolved_uid, resolved_aid) - - meta = inp.get("metadata") or {} - - if "confidence" not in meta: - meta["confidence"] = 0.7 - changed = True - if "files" not in meta: - meta["files"] = ["*"] - changed = True - if "source" not in meta: - meta["source"] = "auto_capture" - changed = True - if "type" not in meta: - meta["type"] = "task_learning" - changed = True - - if meta.get("confidence", 0) >= 1.0 and "infer" not in inp: - inp["infer"] = False - changed = True - - # Track session in metadata instead of run_id. - # run_id creates a separate entity partition in the v3 API, - # making memories invisible to search/get_memories calls - # that don't include a run_id filter. - if "session_id" not in meta: - sid = os.environ.get("MEM0_SESSION_ID", "") - if not sid: - session_file = "/tmp/mem0_session_id_" + os.environ.get("USER", "default") - if os.path.isfile(session_file): - try: - with open(session_file) as f: - sid = f.read().strip() - except OSError: - pass - if sid: - meta["session_id"] = sid - changed = True - - if changed: - inp["metadata"] = meta - -elif handler in ("search_memories", "get_memories"): - if global_search: - inp["filters"] = {"OR": [{"user_id": "*"}]} - changed = True - else: - changed = inject_filter_identity(inp, resolved_uid, resolved_aid) - -elif handler == "delete_all": - changed = inject_top_level_identity(inp, resolved_uid, resolved_aid) - -if changed: - print(json.dumps(inp)) -PYEOF -PATCHED=$(cat "$_PATCH_OUT" 2>/dev/null) -rm -f "$_PATCH_OUT" - -if [ -n "$PATCHED" ] && echo "$PATCHED" | jq empty 2>/dev/null; then - jq -n --argjson updated "$PATCHED" '{ - "hookSpecificOutput": { - "hookEventName": "PreToolUse", - "permissionDecision": "allow", - "updatedInput": $updated - } - }' 2>/dev/null || true -fi - -# Track session stats here because PostToolUse hooks don't fire for plugin MCP tools. -case "$HANDLER" in - add_memory) - _CAT=$(echo "$TOOL_INPUT" | jq -r '.metadata.type // .metadata.category // ""' 2>/dev/null || echo "") - python3 "$SCRIPT_DIR/session_stats.py" add "$_CAT" 2>/dev/null & - ;; - search_memories|get_memories) - python3 "$SCRIPT_DIR/session_stats.py" search 2>/dev/null & - ;; -esac - -exit 0 diff --git a/integrations/mem0-plugin/scripts/ensure_deps.sh b/integrations/mem0-plugin/scripts/ensure_deps.sh deleted file mode 100755 index 8bb54a483..000000000 --- a/integrations/mem0-plugin/scripts/ensure_deps.sh +++ /dev/null @@ -1,50 +0,0 @@ -#!/usr/bin/env bash -# Install mem0ai SDK into a persistent venv inside CLAUDE_PLUGIN_DATA. -# Runs on SessionStart; skips if requirements.txt hasn't changed. -set -euo pipefail - -PLUGIN_ROOT="${CLAUDE_PLUGIN_ROOT:-$(cd "$(dirname "$0")/.." && pwd)}" -DATA_DIR="${CLAUDE_PLUGIN_DATA:-${HOME}/.mem0/plugin-data}" -VENV_DIR="${DATA_DIR}/venv" -REQ_SRC="${PLUGIN_ROOT}/requirements.txt" -REQ_STAMP="${DATA_DIR}/requirements.txt" - -mkdir -p "${DATA_DIR}" - -LOCKDIR="${DATA_DIR}/.install-lock" - -needs_install=false - -if [ ! -f "${VENV_DIR}/bin/python3" ]; then - needs_install=true -elif ! diff -q "${REQ_SRC}" "${REQ_STAMP}" >/dev/null 2>&1; then - needs_install=true -fi - -if [ "${needs_install}" = "true" ]; then - if mkdir "${LOCKDIR}" 2>/dev/null; then - # We acquired the lock — proceed with installation - trap 'rmdir "${LOCKDIR}" 2>/dev/null || true' EXIT - python3 -m venv "${VENV_DIR}" 2>/dev/null || python -m venv "${VENV_DIR}" - "${VENV_DIR}/bin/pip" install --quiet --upgrade pip >/dev/null 2>&1 || true - if "${VENV_DIR}/bin/pip" install --quiet -r "${REQ_SRC}" 2>/dev/null; then - cp "${REQ_SRC}" "${REQ_STAMP}" - rm -f "${DATA_DIR}/.install-failed" - else - rm -f "${REQ_STAMP}" - touch "${DATA_DIR}/.install-failed" - echo "mem0 plugin: failed to install Python dependencies" >&2 - exit 0 - fi - else - # Another process holds the lock — wait up to 60s for it to finish - for i in $(seq 1 60); do - [ ! -d "${LOCKDIR}" ] && break - sleep 1 - done - # Check if the other process's install failed - if [ -f "${DATA_DIR}/.install-failed" ]; then - echo "mem0 plugin: dependency installation failed (by another session)" >&2 - fi - fi -fi diff --git a/integrations/mem0-plugin/scripts/file_context.py b/integrations/mem0-plugin/scripts/file_context.py deleted file mode 100644 index a1ebf2735..000000000 --- a/integrations/mem0-plugin/scripts/file_context.py +++ /dev/null @@ -1,134 +0,0 @@ -#!/usr/bin/env python3 -"""File-context injection for PreToolUse/Read hook. - -When Claude is about to read a file, this script searches mem0 for -memories that reference that file path and returns a compact timeline -of prior work. This gives Claude context like "last time you fixed a -null pointer here" before it reads the file. - -Modeled after claude-mem's file-context handler but adapted for mem0's -cloud API architecture. - -Input: file_path (positional arg), env vars for identity -Output: JSON to stdout with hookSpecificOutput.additionalContext -""" - -from __future__ import annotations - -import os -import sys -from pathlib import Path - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _formatting import TYPE_ICONS, format_age -from _identity import resolve_api_key, resolve_user_id -from _project import resolve_project_id -from _search import search_memories, should_rerank - -FILE_READ_GATE_MIN_BYTES = 1500 -MAX_RESULTS = 5 -SEARCH_TIMEOUT = 5 - - -def gate_file(file_path: str, cwd: str) -> str | None: - """Return the resolved absolute path if the file passes gating, else None.""" - if not file_path: - return None - p = Path(file_path) - if not p.is_absolute(): - p = Path(cwd) / p - try: - p = p.resolve() - if not p.is_file(): - return None - if p.stat().st_size < FILE_READ_GATE_MIN_BYTES: - return None - return str(p) - except OSError: - return None - - -def relative_path(abs_path: str, cwd: str) -> str: - try: - return os.path.relpath(abs_path, cwd) - except ValueError: - return abs_path - - -def format_timeline(memories: list[dict], file_path: str) -> str: - """Format memories into a compact timeline for context injection.""" - if not memories: - return "" - - rel = file_path - lines = [ - f"Prior work on `{rel}` — {len(memories)} memories found.", - "Need details? Use `search_memories` with the memory ID.", - "", - ] - - for m in memories: - mid = m.get("id", "?")[:8] - text = (m.get("memory", "") or "")[:150].replace("\n", " ").strip() - meta = m.get("metadata") or {} - cat = meta.get("type", "unknown") - icon = TYPE_ICONS.get(cat, "❓") - age = format_age(m) - age_str = f" ({age})" if age else "" - lines.append(f"- {icon} [{cat}]{age_str} {text} [mem0:{mid}]") - - return "\n".join(lines) - - -def search_file_context( - api_key: str, user_id: str, project_id: str, file_path: str, cwd: str -) -> str: - """Search mem0 for memories related to a file path.""" - global_search = os.environ.get("MEM0_GLOBAL_SEARCH", "false") == "true" - rel = relative_path(file_path, cwd) - basename = os.path.basename(file_path) - - query = f"{rel} {basename}" if rel != basename else rel - results = search_memories( - api_key, user_id, project_id, query, - top_k=MAX_RESULTS, threshold=0.3, - global_search=global_search, - rerank=should_rerank(), - ) - - results = results[:MAX_RESULTS] - - return format_timeline(results, rel) - - -def main(): - if len(sys.argv) < 2: - sys.exit(0) - - file_path = sys.argv[1] - cwd = sys.argv[2] if len(sys.argv) > 2 else os.getcwd() - - api_key = resolve_api_key() - if not api_key: - sys.exit(0) - - resolved = gate_file(file_path, cwd) - if not resolved: - sys.exit(0) - - user_id = resolve_user_id() - project_id = resolve_project_id(cwd) - - timeline = search_file_context(api_key, user_id, project_id, resolved, cwd) - if not timeline: - sys.exit(0) - - print(timeline, end="") - - -if __name__ == "__main__": - try: - main() - except Exception: - pass - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/import_competing_tools.py b/integrations/mem0-plugin/scripts/import_competing_tools.py deleted file mode 100644 index 968bbbcd2..000000000 --- a/integrations/mem0-plugin/scripts/import_competing_tools.py +++ /dev/null @@ -1,299 +0,0 @@ -#!/usr/bin/env python3 -"""Import memories from competing AI tool configuration files into mem0. - -Sub-commands (via sys.argv[1]): - cursorrules [--path .cursorrules] - copilot [--path .github/copilot-instructions.md] - cline [--path memory-bank/] - continue [--path .continue/rules.md] - -Each sub-command reads configuration files from competing tools, -splits them into chunks, and POSTs each chunk to the mem0 API as a -project_profile memory. - -Output: progress messages to stdout, errors to stderr -Exit: 0 always -""" - -from __future__ import annotations - -import hashlib -import json -import os -import sys -import urllib.error -import urllib.request - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _chunking import ( - filter_and_truncate, - split_by_headers, - split_by_hr_or_headers, -) -from _identity import resolve_api_key, resolve_user_id -from _project import resolve_branch, resolve_project_id - -API_URL = "https://api.mem0.ai" -HASH_STORE = os.path.expanduser("~/.mem0/import_hashes.json") - - -def _load_hashes() -> dict[str, str]: - if not os.path.isfile(HASH_STORE): - return {} - try: - with open(HASH_STORE) as f: - return json.load(f) - except (OSError, json.JSONDecodeError): - return {} - - -def _save_hashes(hashes: dict[str, str]) -> None: - os.makedirs(os.path.dirname(HASH_STORE), exist_ok=True) - try: - with open(HASH_STORE, "w") as f: - json.dump(hashes, f, indent=2) - except OSError: - pass - - -def _content_hash(content: str) -> str: - return hashlib.sha256(content.encode("utf-8")).hexdigest() - - -# --------------------------------------------------------------------------- -# API helpers -# --------------------------------------------------------------------------- - - -def post_memory(api_key: str, content: str, user_id: str, project_id: str, branch: str, source: str) -> bool: - """POST a single memory chunk to the mem0 API.""" - metadata: dict = { - "type": "project_profile", - "source": source, - } - if branch: - metadata["branch"] = branch - - body = { - "messages": [{"role": "user", "content": content}], - "user_id": user_id, - "app_id": project_id, - "metadata": metadata, - "infer": False, - } - data = json.dumps(body).encode("utf-8") - req = urllib.request.Request( - f"{API_URL}/v3/memories/add/", - data=data, - headers={ - "Content-Type": "application/json", - "Authorization": f"Token {api_key}", - }, - method="POST", - ) - try: - with urllib.request.urlopen(req, timeout=20) as resp: - return resp.status in (200, 201) - except urllib.error.URLError as e: - print(f" [warn] API call failed: {e}", file=sys.stderr) - return False - - -def import_chunks(chunks: list[str], api_key: str, user_id: str, project_id: str, branch: str, source: str, hash_key: str = "") -> int: - """Import a list of content chunks; return number of successful imports. - - Skips import if content hash matches a previous run for the same hash_key.""" - if hash_key: - combined = "\n".join(chunks) - current_hash = _content_hash(combined) - hashes = _load_hashes() - if hashes.get(hash_key) == current_hash: - print(f"Already imported (unchanged) -- skipping: {hash_key}") - return 0 - else: - current_hash = "" - hashes = {} - - success = 0 - for chunk in chunks: - if post_memory(api_key, chunk, user_id, project_id, branch, source): - success += 1 - - if success > 0 and hash_key and current_hash: - hashes[hash_key] = current_hash - _save_hashes(hashes) - - return success - - -# --------------------------------------------------------------------------- -# Sub-command implementations -# --------------------------------------------------------------------------- - - -def _parse_path_arg(args: list[str], flag: str, default: str) -> str: - """Extract --path from args list, falling back to default.""" - for i, arg in enumerate(args): - if arg == flag and i + 1 < len(args): - return args[i + 1] - if arg.startswith(f"{flag}="): - return arg[len(flag) + 1:] - return default - - -def cmd_cursorrules(args: list[str]) -> None: - path = _parse_path_arg(args, "--path", ".cursorrules") - source = "cursor-import" - - api_key = resolve_api_key() - user_id = resolve_user_id() - project_id = resolve_project_id() - branch = resolve_branch() - - if not api_key: - print("Error: MEM0_API_KEY not set", file=sys.stderr) - return - - if not os.path.isfile(path): - print(f"File not found: {path}", file=sys.stderr) - return - - with open(path, encoding="utf-8", errors="replace") as f: - content = f.read() - - raw_chunks = split_by_headers(content, "## ") - # Fall back to treating the whole file as one chunk if no headers found - if not raw_chunks: - raw_chunks = [content.strip()] if content.strip() else [] - - chunks = filter_and_truncate(raw_chunks) - n = import_chunks(chunks, api_key, user_id, project_id, branch, source, hash_key=f"{project_id}:{source}:{path}") - print(f"Imported {n} memories from {source} ({path})") - - -def cmd_copilot(args: list[str]) -> None: - path = _parse_path_arg(args, "--path", ".github/copilot-instructions.md") - source = "copilot-import" - - api_key = resolve_api_key() - user_id = resolve_user_id() - project_id = resolve_project_id() - branch = resolve_branch() - - if not api_key: - print("Error: MEM0_API_KEY not set", file=sys.stderr) - return - - if not os.path.isfile(path): - print(f"File not found: {path}", file=sys.stderr) - return - - with open(path, encoding="utf-8", errors="replace") as f: - content = f.read() - - raw_chunks = split_by_headers(content, "## ") - if not raw_chunks: - raw_chunks = [content.strip()] if content.strip() else [] - - chunks = filter_and_truncate(raw_chunks) - n = import_chunks(chunks, api_key, user_id, project_id, branch, source, hash_key=f"{project_id}:{source}:{path}") - print(f"Imported {n} memories from {source} ({path})") - - -def cmd_cline(args: list[str]) -> None: - dir_path = _parse_path_arg(args, "--path", "memory-bank/") - source = "cline-import" - - api_key = resolve_api_key() - user_id = resolve_user_id() - project_id = resolve_project_id() - branch = resolve_branch() - - if not api_key: - print("Error: MEM0_API_KEY not set", file=sys.stderr) - return - - if not os.path.isdir(dir_path): - print(f"Directory not found: {dir_path}", file=sys.stderr) - return - - md_files = sorted( - f for f in os.listdir(dir_path) if f.endswith(".md") - ) - if not md_files: - print(f"No .md files found in {dir_path}", file=sys.stderr) - return - - total = 0 - for filename in md_files: - filepath = os.path.join(dir_path, filename) - with open(filepath, encoding="utf-8", errors="replace") as f: - content = f.read().strip() - if not content: - continue - chunks = filter_and_truncate([content]) - n = import_chunks(chunks, api_key, user_id, project_id, branch, source, hash_key=f"{project_id}:{source}:{filepath}") - total += n - - print(f"Imported {total} memories from {source} ({dir_path})") - - -def cmd_continue(args: list[str]) -> None: - path = _parse_path_arg(args, "--path", ".continue/rules.md") - source = "continue-import" - - api_key = resolve_api_key() - user_id = resolve_user_id() - project_id = resolve_project_id() - branch = resolve_branch() - - if not api_key: - print("Error: MEM0_API_KEY not set", file=sys.stderr) - return - - if not os.path.isfile(path): - print(f"File not found: {path}", file=sys.stderr) - return - - with open(path, encoding="utf-8", errors="replace") as f: - content = f.read() - - raw_chunks = split_by_hr_or_headers(content) - if not raw_chunks: - raw_chunks = [content.strip()] if content.strip() else [] - - chunks = filter_and_truncate(raw_chunks) - n = import_chunks(chunks, api_key, user_id, project_id, branch, source, hash_key=f"{project_id}:{source}:{path}") - print(f"Imported {n} memories from {source} ({path})") - - -# --------------------------------------------------------------------------- -# Entry point -# --------------------------------------------------------------------------- - -COMMANDS = { - "cursorrules": cmd_cursorrules, - "copilot": cmd_copilot, - "cline": cmd_cline, - "continue": cmd_continue, -} - - -def main() -> None: - if len(sys.argv) < 2 or sys.argv[1] not in COMMANDS: - available = ", ".join(COMMANDS.keys()) - print("Usage: import_competing_tools.py [--path ]", file=sys.stderr) - print(f"Subcommands: {available}", file=sys.stderr) - sys.exit(0) - - subcommand = sys.argv[1] - remaining_args = sys.argv[2:] - COMMANDS[subcommand](remaining_args) - - -if __name__ == "__main__": - try: - main() - except Exception as e: - print(f"Unexpected error: {e}", file=sys.stderr) - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/install_codex_hooks.py b/integrations/mem0-plugin/scripts/install_codex_hooks.py deleted file mode 100755 index c94ef1561..000000000 --- a/integrations/mem0-plugin/scripts/install_codex_hooks.py +++ /dev/null @@ -1,162 +0,0 @@ -#!/usr/bin/env python3 -"""Install Mem0 lifecycle hooks into ~/.codex/hooks.json. - -Codex discovers hooks only at ~/.codex/hooks.json or /.codex/hooks.json, -and has no plugin-host mechanism for auto-wiring hooks from an installed -plugin. This installer reads the template at hooks/codex-hooks.json, rewrites -the ${PLUGIN_ROOT} placeholder to the absolute install path of this -plugin, then merges the entries into ~/.codex/hooks.json. - -Re-running is idempotent: existing Mem0 entries (identified by the plugin -directory name in the command string) are removed before fresh entries are -added, so upgrades don't leave duplicates. - -Usage: - python3 install_codex_hooks.py # install or update - python3 install_codex_hooks.py --uninstall # remove Mem0 entries - -After installing, Codex requires the hooks feature flag in ~/.codex/config.toml: - - [features] - codex_hooks = true -""" - -from __future__ import annotations - -import argparse -import json -import platform -import sys -from pathlib import Path - -SCRIPT_DIR = Path(__file__).resolve().parent -PLUGIN_ROOT = SCRIPT_DIR.parent - -CODEX_DIR = Path.home() / ".codex" -HOOKS_FILE = CODEX_DIR / "hooks.json" -CONFIG_FILE = CODEX_DIR / "config.toml" - -TEMPLATE_FILE = PLUGIN_ROOT / "hooks" / "codex-hooks.json" - -# Substring we look for when identifying entries this installer owns. -# Matches the plugin directory name, which stays stable across install paths. -OWNER_MARKER = "mem0-plugin" - - -def load_template() -> dict: - raw = TEMPLATE_FILE.read_text() - raw = raw.replace("${PLUGIN_ROOT}", str(PLUGIN_ROOT)) - return json.loads(raw) - - -def load_existing() -> dict: - if not HOOKS_FILE.exists(): - return {"hooks": {}} - try: - return json.loads(HOOKS_FILE.read_text()) - except (json.JSONDecodeError, OSError) as e: - print(f"error: failed to read {HOOKS_FILE}: {e}", file=sys.stderr) - sys.exit(1) - - -def is_owned_entry(entry: dict) -> bool: - for hook in entry.get("hooks", []): - if OWNER_MARKER in hook.get("command", ""): - return True - return False - - -def strip_owned_entries(config: dict) -> dict: - hooks = config.get("hooks", {}) or {} - for event in list(hooks.keys()): - hooks[event] = [e for e in hooks[event] if not is_owned_entry(e)] - if not hooks[event]: - del hooks[event] - config["hooks"] = hooks - return config - - -def merge_template(config: dict, template: dict) -> dict: - hooks = config.setdefault("hooks", {}) - for event, entries in template.get("hooks", {}).items(): - hooks.setdefault(event, []).extend(entries) - return config - - -def write_config(config: dict) -> None: - CODEX_DIR.mkdir(parents=True, exist_ok=True) - HOOKS_FILE.write_text(json.dumps(config, indent=2) + "\n") - - -def feature_flag_enabled() -> bool: - if not CONFIG_FILE.exists(): - return False - content = CONFIG_FILE.read_text() - for line in content.splitlines(): - stripped = line.split("#", 1)[0].strip().replace(" ", "") - if stripped == "codex_hooks=true": - return True - return False - - -def print_feature_flag_hint() -> None: - print() - print("Codex hooks feature flag is not enabled.") - print(f"Add this to {CONFIG_FILE}:") - print() - print(" [features]") - print(" codex_hooks = true") - print() - print("Then restart Codex.") - - -def main() -> int: - parser = argparse.ArgumentParser(description="Install or remove Mem0 Codex hooks.") - parser.add_argument( - "--uninstall", - action="store_true", - help="Remove Mem0 entries from ~/.codex/hooks.json and exit.", - ) - args = parser.parse_args() - - config = load_existing() - - if args.uninstall: - config = strip_owned_entries(config) - write_config(config) - print(f"Removed Mem0 hooks from {HOOKS_FILE}") - return 0 - - # Codex lifecycle hooks register .sh paths directly in ~/.codex/hooks.json. - # On native Windows .sh has no default handler, so Codex spawning a hook - # triggers "Open With" dialogs (one OpenWith.exe per event). See #5243. - if platform.system() == "Windows": - print( - "Codex lifecycle hooks register .sh scripts directly, which Windows\n" - "cannot execute without a bash interpreter on PATH. Re-run this\n" - "installer from WSL or Git Bash, or use Mem0 via MCP / Direct tools\n", - file=sys.stderr, - ) - return 2 - - if not TEMPLATE_FILE.exists(): - print(f"error: template not found at {TEMPLATE_FILE}", file=sys.stderr) - return 1 - - template = load_template() - config = strip_owned_entries(config) - config = merge_template(config, template) - write_config(config) - - print(f"Installed Mem0 hooks into {HOOKS_FILE}") - print(f"Plugin path: {PLUGIN_ROOT}") - print("Events: PreToolUse, SessionStart, UserPromptSubmit, PostToolUse, Stop, PreCompact") - - if not feature_flag_enabled(): - print_feature_flag_hint() - - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/integrations/mem0-plugin/scripts/kimi_hook_shim.sh b/integrations/mem0-plugin/scripts/kimi_hook_shim.sh deleted file mode 100755 index 24e622706..000000000 --- a/integrations/mem0-plugin/scripts/kimi_hook_shim.sh +++ /dev/null @@ -1,120 +0,0 @@ -#!/usr/bin/env bash -# Kimi Code hook adapter. -# -# Kimi Code's hook contract is close to Claude Code's but differs in four ways -# that the mem0 hook scripts care about. Rather than fork nine scripts, every -# Kimi hook entry in .kimi-plugin/plugin.json routes through this shim. -# -# 1. cwd Kimi runs hook commands with cwd forced to the PLUGIN root -# (agent-core-v2 src/app/plugin/manager.ts -> enabledHooks() sets -# `cwd: record.root`). The real project directory only arrives as -# the payload's `cwd` field. _project.sh resolves MEM0_PROJECT_ID -# from $PWD/`git remote`, so we chdir into the payload cwd first. -# -# 2. stdin Kimi sends snake_case JSON like Claude, but: -# - `prompt` is a ContentPart[] array, not a string -# - PostToolUse sends `tool_output`, not `tool_response` -# - file tools use `tool_input.path`, not `tool_input.file_path` -# - plugin MCP tools are `mcp__plugin-mem0_mem0__*` (hyphen), -# not Claude's `mcp__plugin_mem0_mem0__*` -# - there is NO `transcript_path` (no equivalent exists) -# We normalise the first four into the Claude shape. -# -# 3. stdout Kimi's hook stdout parser (agent-core-v2 -# src/agent/externalHooks/runner.ts -> HookJsonOutputSchema) only -# understands top-level `message`, `hookSpecificOutput.message`, -# `hookSpecificOutput.permissionDecision` and -# `hookSpecificOutput.permissionDecisionReason`. -# `additionalContext` and `updatedInput` are NOT recognised. -# Raw (non-JSON) stdout IS appended to context for UserPromptSubmit -# (user-prompt.ts -> userPromptHookMessage falls back to stdout), so -# we unwrap additionalContext into plain text. -# -# 4. exit Same as Claude: 0 = allow, 2 = block (stderr is the reason), -# any other code / timeout = fail-open. -# -# Usage: kimi_hook_shim.sh [args...] - -set -uo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" - -[ $# -ge 1 ] || exit 0 -TARGET_NAME="$1" -shift -TARGET="$SCRIPT_DIR/$TARGET_NAME" -[ -f "$TARGET" ] || exit 0 - -# Telemetry attribution (telemetry.py::detect_platform honours MEM0_PLATFORM). -export MEM0_PLATFORM="${MEM0_PLATFORM:-kimi}" - -RAW=$(cat) - -_TMP_BASE="${TMPDIR:-/tmp}" -IN_FILE="$_TMP_BASE/mem0_kimi_in_$$" -OUT_FILE="$_TMP_BASE/mem0_kimi_out_$$" -trap 'rm -f "$IN_FILE" "$OUT_FILE"' EXIT - -HAVE_JQ="" -command -v jq >/dev/null 2>&1 && HAVE_JQ="true" - -# ---------------------------------------------------------------- stdin shape -NORM="" -if [ -n "$HAVE_JQ" ]; then - NORM=$(printf '%s' "$RAW" | jq -c ' - def flat_prompt: - if type == "array" then - [ .[]? | if type == "object" then (.text // "") elif type == "string" then . else "" end ] - | map(select(. != "")) | join("\n") - elif type == "string" then . - else "" end; - . as $in - | (if has("prompt") then .prompt = ($in.prompt | flat_prompt) else . end) - | (if (has("tool_response") | not) and (.tool_output != null) - then .tool_response = .tool_output else . end) - | (if ((.tool_input | type) == "object") and (.tool_input.file_path == null) and (.tool_input.path != null) - then .tool_input.file_path = .tool_input.path else . end) - | (if ((.tool_name | type) == "string") and (.tool_name | startswith("mcp__")) and (.tool_name | test("mem0")) - then .tool_name = (.tool_name | gsub("-"; "_")) else . end) - ' 2>/dev/null) -fi -[ -n "$NORM" ] || NORM="$RAW" - -printf '%s' "$NORM" >"$IN_FILE" 2>/dev/null || exit 0 - -# ------------------------------------------------------------------- real cwd -if [ -n "$HAVE_JQ" ]; then - PROJECT_CWD=$(printf '%s' "$NORM" | jq -r '.cwd // ""' 2>/dev/null || printf '') - if [ -n "$PROJECT_CWD" ] && [ -d "$PROJECT_CWD" ]; then - cd "$PROJECT_CWD" 2>/dev/null || true - fi -fi - -# ------------------------------------------------------------------- dispatch -# Redirect (not pipe) stdout so backgrounded children inside the hook scripts -# cannot hold the shim open until Kimi's timeout fires. -bash "$TARGET" "$@" <"$IN_FILE" >"$OUT_FILE" -CODE=$? - -OUT=$(cat "$OUT_FILE" 2>/dev/null || printf '') - -# --------------------------------------------------------------- stdout shape -if [ -n "$OUT" ] && [ -n "$HAVE_JQ" ] && printf '%s' "$OUT" | jq -e 'type == "object"' >/dev/null 2>&1; then - if printf '%s' "$OUT" | jq -e '.hookSpecificOutput.permissionDecision == "deny"' >/dev/null 2>&1; then - # Kimi understands deny natively — pass the envelope straight through. - printf '%s' "$OUT" - else - CTX=$(printf '%s' "$OUT" | jq -r ' - (.hookSpecificOutput.additionalContext - // .additionalContext - // .message - // .hookSpecificOutput.message - // "") - | gsub("\\\\n"; "\n")' 2>/dev/null || printf '') - [ -n "$CTX" ] && printf '%s\n' "$CTX" - fi -elif [ -n "$OUT" ]; then - printf '%s' "$OUT" -fi - -exit $CODE diff --git a/integrations/mem0-plugin/scripts/load_settings.py b/integrations/mem0-plugin/scripts/load_settings.py deleted file mode 100644 index 139d180c6..000000000 --- a/integrations/mem0-plugin/scripts/load_settings.py +++ /dev/null @@ -1,71 +0,0 @@ -"""Load plugin settings from ~/.mem0/settings.json. - -Settings file is user-editable. Missing file or keys fall back to defaults. -""" - -from __future__ import annotations - -import json -from pathlib import Path - -SETTINGS_PATH = Path.home() / ".mem0" / "settings.json" - -DEFAULTS = { - "auto_save": True, - "auto_search": True, - "search_limit": 10, - "retention_session_days": 90, - "confidence_threshold": 0.3, - "global_search": False, - "debug": False, -} - - -def load_settings() -> dict: - settings = dict(DEFAULTS) - if SETTINGS_PATH.exists(): - try: - with open(SETTINGS_PATH) as f: - user = json.load(f) - except (json.JSONDecodeError, OSError): - return settings - if isinstance(user, dict): - settings.update({k: v for k, v in user.items() if k in DEFAULTS}) - return settings - - -def create_default_settings() -> bool: - """Write the default settings file if absent. Returns True if it was created.""" - SETTINGS_PATH.parent.mkdir(parents=True, exist_ok=True) - if SETTINGS_PATH.exists(): - return False - with open(SETTINGS_PATH, "w") as f: - json.dump(DEFAULTS, f, indent=2) - f.write("\n") - return True - - -def unknown_keys() -> list[str]: - """Keys present in the user's settings file that no code reads.""" - if not SETTINGS_PATH.exists(): - return [] - try: - with open(SETTINGS_PATH) as f: - user = json.load(f) - except (json.JSONDecodeError, OSError): - return [] - if not isinstance(user, dict): - return [] - return sorted(k for k in user if k not in DEFAULTS) - - -if __name__ == "__main__": - import sys - if len(sys.argv) > 1 and sys.argv[1] == "init": - if create_default_settings(): - print(f"Created {SETTINGS_PATH}") - ignored = unknown_keys() - if ignored: - print(f"Ignoring unrecognized settings in {SETTINGS_PATH}: {', '.join(ignored)}") - else: - print(json.dumps(load_settings())) diff --git a/integrations/mem0-plugin/scripts/on_bash_output.sh b/integrations/mem0-plugin/scripts/on_bash_output.sh deleted file mode 100755 index dc8621fc9..000000000 --- a/integrations/mem0-plugin/scripts/on_bash_output.sh +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env bash -# Hook: PostToolUse (matcher: Bash) -# -# Scans bash command output for stack traces and error patterns. -# When found, injects a search rubric telling the agent to check mem0 -# for prior occurrences of the same error. -# -# This complements on_user_prompt.sh (which catches errors in the user's -# typed message). This hook catches errors in COMMAND OUTPUT — e.g., -# when `npm test` or `python script.py` fails with a traceback. -# -# Input: JSON on stdin with tool_name, tool_input, tool_response -# Output: Context injected into Claude's next response (exit 0) - -set -uo pipefail - -INPUT=$(cat) - -TOOL_RESULT=$(echo "$INPUT" | jq -r '.tool_response // ""' 2>/dev/null || echo "") - -# Skip short output (< 50 chars unlikely to contain a real stack trace) -if [ ${#TOOL_RESULT} -lt 50 ]; then - exit 0 -fi - -# Skip git operations — not useful for error detection -COMMAND=$(echo "$INPUT" | jq -r '.tool_input.command // ""' 2>/dev/null || echo "") -case "$COMMAND" in - *"git commit"*|*"git merge"*|*"git rebase"*) - exit 0 - ;; -esac - -# Detect stack traces and error patterns in command output -HAS_ERROR="" -if echo "$TOOL_RESULT" | grep -qE '(Traceback \(most recent call last\)|panic: |FATAL:|error\[E[0-9]+\])'; then - HAS_ERROR="true" -elif [ "$(echo "$TOOL_RESULT" | grep -cE '(Error:|Exception:)')" -ge 2 ]; then - HAS_ERROR="true" -fi - -if [ -z "$HAS_ERROR" ]; then - exit 0 -fi - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -. "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true - -# Extract error class/message (first matching line) -ERROR_LINE=$(echo "$TOOL_RESULT" | grep -iE '(Error:|Exception:|panic:|FAIL:|fatal:)' | head -1 | sed 's/^[[:space:]]*//' | cut -c1-120) - -# Extract file paths from stack trace frames -TRACE_FILES=$(echo "$TOOL_RESULT" | grep -oE '([a-zA-Z0-9_./-]+\.(py|ts|tsx|js|jsx|rs|go|rb|java|sh))(:[0-9]+)?' | head -5 | sort -u) - -# Build file list for display -FILE_DISPLAY="" -if [ -n "$TRACE_FILES" ]; then - FILE_DISPLAY=$(echo "$TRACE_FILES" | sed 's/^/ - /') -fi - -USER_ID="${MEM0_RESOLVED_USER_ID:-${USER:-default}}" - -# Extract query (first 80 chars of error line) -ERROR_QUERY=$(echo "$ERROR_LINE" | cut -c1-80) - -# Telemetry (fire regardless of API key) -python3 "$SCRIPT_DIR/telemetry.py" bash_error --error_detected 2>/dev/null & - -# No API key — skip output entirely -if [ -z "${MEM0_API_KEY:-}" ]; then - exit 0 -fi - -# Pre-fetch memories: anti_pattern and bug_fix searches -RESULTS=$(PYTHONPATH="$SCRIPT_DIR" MEM0_SEARCH_QUERY="$ERROR_QUERY" MEM0_SEARCH_USER="$USER_ID" \ - MEM0_API_KEY="${MEM0_API_KEY}" MEM0_PROJECT_ID="${MEM0_PROJECT_ID:-unknown}" \ - python3 -c " -import os, sys -sys.path.insert(0, os.environ.get('PYTHONPATH', '.')) -from _search import search_memories, format_results_for_context, should_rerank - -api_key = os.environ.get('MEM0_API_KEY', '') -user_id = os.environ.get('MEM0_SEARCH_USER', 'default') -project_id = os.environ.get('MEM0_PROJECT_ID', 'unknown') -query = os.environ.get('MEM0_SEARCH_QUERY', '') -rerank = should_rerank() - -r1 = search_memories(api_key, user_id, project_id, query, metadata_type='anti_pattern', top_k=3, rerank=rerank) -r2 = search_memories(api_key, user_id, project_id, query, metadata_type='bug_fix', top_k=3, rerank=rerank) - -seen = set() -combined = [] -for m in r1 + r2: - mid = m.get('id', '') - if mid not in seen: - seen.add(mid) - combined.append(m) - -print(format_results_for_context(combined, heading='Prior error memories'), end='') -" 2>/dev/null || echo "") - -# Build context string for JSON output -CTX="Error detected in command output\n\n" -CTX="${CTX}\`${COMMAND}\` produced an error:\n> ${ERROR_LINE}\n" - -if [ -n "$FILE_DISPLAY" ]; then - CTX="${CTX}\nFiles in stack trace:\n${FILE_DISPLAY}\n" -fi - -if [ -n "$RESULTS" ]; then - CTX="${CTX}\n${RESULTS}\n" -fi - -CTX="${CTX}\nResolved errors are stored as anti_pattern or bug_fix memories for future reference." - -jq -cn --arg ctx "$CTX" '{ - hookSpecificOutput: { - hookEventName: "PostToolUse", - additionalContext: $ctx - } -}' 2>/dev/null || true - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_file_read.sh b/integrations/mem0-plugin/scripts/on_file_read.sh deleted file mode 100755 index d190d81d9..000000000 --- a/integrations/mem0-plugin/scripts/on_file_read.sh +++ /dev/null @@ -1,51 +0,0 @@ -#!/usr/bin/env bash -# Hook: PreToolUse (matcher: Read) -# -# Injects prior work context before Claude reads a file. Searches mem0 -# for memories referencing the file path and returns a compact timeline. -# -# Modeled after claude-mem's file-context handler, adapted for mem0 cloud API. -# -# Input: JSON on stdin with tool_name, tool_input (file_path), cwd -# Output: JSON with hookSpecificOutput.additionalContext + permissionDecision -# -# Must never block the Read — silent exit on any failure. - -set -uo pipefail - -INPUT=$(cat) - -# Extract file path from tool_input -FILE_PATH=$(echo "$INPUT" | jq -r '.tool_input.file_path // ""' 2>/dev/null || echo "") -if [ -z "$FILE_PATH" ]; then - exit 0 -fi - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" - -# Resolve API key (covers Desktop app users who set it in shell profile) -if [ -z "${MEM0_API_KEY:-}" ]; then - . "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true -fi -if [ -z "${MEM0_API_KEY:-}" ]; then - exit 0 -fi -CWD=$(echo "$INPUT" | jq -r '.cwd // "."' 2>/dev/null || echo ".") - -# Call the Python worker — it handles gating (file size, existence) -TIMELINE=$(python3 "$SCRIPT_DIR/file_context.py" "$FILE_PATH" "$CWD" 2>/dev/null || echo "") - -if [ -z "$TIMELINE" ]; then - exit 0 -fi - -# Return context injection with permissionDecision: allow -jq -cn --arg ctx "$TIMELINE" '{ - hookSpecificOutput: { - hookEventName: "PreToolUse", - additionalContext: $ctx, - permissionDecision: "allow" - } -}' 2>/dev/null || true - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_file_read_cursor.sh b/integrations/mem0-plugin/scripts/on_file_read_cursor.sh deleted file mode 100755 index c0531d4e1..000000000 --- a/integrations/mem0-plugin/scripts/on_file_read_cursor.sh +++ /dev/null @@ -1,41 +0,0 @@ -#!/usr/bin/env bash -# Hook: preToolUse (matcher: Read) — Cursor variant -# -# Same as on_file_read.sh but uses CURSOR_PLUGIN_ROOT for path resolution -# and sources Cursor-specific identity. - -set -uo pipefail - -INPUT=$(cat) - -FILE_PATH=$(echo "$INPUT" | jq -r '.tool_input.file_path // ""' 2>/dev/null || echo "") -if [ -z "$FILE_PATH" ]; then - exit 0 -fi - -if [ -z "${MEM0_API_KEY:-}" ]; then - SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" - . "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true -fi -if [ -z "${MEM0_API_KEY:-}" ]; then - exit 0 -fi - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -CWD=$(echo "$INPUT" | jq -r '.cwd // "."' 2>/dev/null || echo ".") - -TIMELINE=$(python3 "$SCRIPT_DIR/file_context.py" "$FILE_PATH" "$CWD" 2>/dev/null || echo "") - -if [ -z "$TIMELINE" ]; then - exit 0 -fi - -jq -cn --arg ctx "$TIMELINE" '{ - hookSpecificOutput: { - hookEventName: "PreToolUse", - additionalContext: $ctx, - permissionDecision: "allow" - } -}' 2>/dev/null || true - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_post_tool_use.sh b/integrations/mem0-plugin/scripts/on_post_tool_use.sh deleted file mode 100755 index 55636cfa5..000000000 --- a/integrations/mem0-plugin/scripts/on_post_tool_use.sh +++ /dev/null @@ -1,36 +0,0 @@ -#!/usr/bin/env bash -# Hook: PostToolUse — track mem0 MCP tool usage for session stats -# -# Fires after any tool call. We only care about mem0 MCP tools: -# mcp__mem0__add_memory → record an add -# mcp__mem0__search_memories → record a search -# -# Input: JSON on stdin with tool_name, tool_input, tool_response -# Output: none (exit 0, non-blocking) - -set -uo pipefail - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" - -INPUT=$(cat) -TOOL_NAME=$(echo "$INPUT" | jq -r '.tool_name // ""' 2>/dev/null || echo "") - -case "$TOOL_NAME" in - *__add_memory) - CATEGORY=$(echo "$INPUT" | jq -r '.tool_input.metadata.type // .tool_input.metadata.category // ""' 2>/dev/null || echo "") - python3 "$SCRIPT_DIR/session_stats.py" add "$CATEGORY" 2>/dev/null || true - python3 "$SCRIPT_DIR/telemetry.py" tool_use --tool=add_memory 2>/dev/null & - ;; - *__search_memories|*__get_memories) - python3 "$SCRIPT_DIR/session_stats.py" search 2>/dev/null || true - python3 "$SCRIPT_DIR/telemetry.py" tool_use --tool=search_memories 2>/dev/null & - ;; - *__delete_memory) - python3 "$SCRIPT_DIR/telemetry.py" tool_use --tool=delete_memory 2>/dev/null & - ;; - *__update_memory) - python3 "$SCRIPT_DIR/telemetry.py" tool_use --tool=update_memory 2>/dev/null & - ;; -esac - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_post_tool_use_cursor.sh b/integrations/mem0-plugin/scripts/on_post_tool_use_cursor.sh deleted file mode 100755 index bcfb39ace..000000000 --- a/integrations/mem0-plugin/scripts/on_post_tool_use_cursor.sh +++ /dev/null @@ -1,18 +0,0 @@ -#!/usr/bin/env bash -# Hook: postToolUse (Cursor) — track mem0 MCP tool usage for session stats -# -# Wraps on_post_tool_use.sh. Cursor expects JSON output but PostToolUse -# has no meaningful return value, so we output {} after tracking. - -set -uo pipefail - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" - -# Pin platform so the shared script's telemetry is attributed to cursor. -export MEM0_PLATFORM=cursor - -# Run the shared tracker (output is ignored) -"$SCRIPT_DIR/on_post_tool_use.sh" 2>/dev/null || true - -echo '{}' -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_pre_compact.py b/integrations/mem0-plugin/scripts/on_pre_compact.py deleted file mode 100755 index 882e26481..000000000 --- a/integrations/mem0-plugin/scripts/on_pre_compact.py +++ /dev/null @@ -1,314 +0,0 @@ -#!/usr/bin/env python3 -"""Capture session state via the Mem0 REST API. - -Safety net for PreCompact and Stop hooks — reads the transcript JSONL, -extracts structured session state, and stores it in Mem0 directly. - -Used by: - - PreCompact hook: Tags with "pre-compaction" (context about to be lost) - - Stop hook: Tags with "session-end" (session ending, Claude can't respond) - -Input: JSON on stdin with transcript_path, session_id, cwd -Output: stderr logs only (exit 0 always — must not block) -""" - -from __future__ import annotations - -import json -import logging -import os -import sys -import urllib.error -import urllib.request -from datetime import date, timedelta - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _identity import resolve_api_key, resolve_user_id -from _project import resolve_branch, resolve_project_id - -log = logging.getLogger("mem0-capture") -log.setLevel(logging.DEBUG) -_handler = logging.StreamHandler(sys.stderr) -_handler.setFormatter(logging.Formatter("[mem0-capture] %(message)s")) -log.addHandler(_handler) - -if os.environ.get("MEM0_DEBUG"): - _log_dir = os.path.expanduser("~/.mem0") - try: - os.makedirs(_log_dir, exist_ok=True) - _file_handler = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) - _file_handler.setFormatter(logging.Formatter("[mem0-capture] %(asctime)s %(message)s")) - log.addHandler(_file_handler) - except OSError: - pass - -API_URL = "https://api.mem0.ai" -MAX_TAIL_LINES = 500 -MAX_USER_MESSAGES = 30 -MAX_BASH_COMMANDS = 20 -MAX_ASSISTANT_TEXT = 10000 -# session_state captures churn fast (active codebase, files in flight). Past -# ~3 months they're stale noise. Durable facts (decisions, conventions) are -# stored separately by the agent without an expiration_date. -SESSION_STATE_EXPIRY_DAYS = 90 - - -def tail_lines(filepath: str, n: int) -> list[str]: - """Read last n lines of a file efficiently.""" - try: - with open(filepath, "rb") as f: - f.seek(0, 2) - file_size = f.tell() - if file_size == 0: - return [] - chunk_size = min(file_size, n * 4096) - f.seek(max(0, file_size - chunk_size)) - data = f.read().decode("utf-8", errors="replace") - return data.splitlines()[-n:] - except OSError: - return [] - - -def parse_transcript(lines: list[str]) -> dict: - """Parse transcript JSONL lines and extract session state.""" - user_messages: list[str] = [] - files_modified: set[str] = set() - bash_commands: list[str] = [] - last_assistant_text = "" - - for line in lines: - line = line.strip() - if not line: - continue - try: - entry = json.loads(line) - except json.JSONDecodeError: - continue - - entry_type = entry.get("type") - if entry_type not in ("user", "assistant"): - continue - if entry.get("isSidechain"): - continue - - message = entry.get("message", {}) - content_blocks = message.get("content", []) - - if entry_type == "user": - parts = [] - if isinstance(content_blocks, str): - parts.append(content_blocks) - elif isinstance(content_blocks, list): - for block in content_blocks: - if isinstance(block, str): - parts.append(block) - elif isinstance(block, dict) and block.get("type") == "text": - parts.append(block.get("text", "")) - text = "\n".join(parts).strip() - if text and len(text) > 10 and not text.startswith("<"): - user_messages.append(text) - - elif entry_type == "assistant": - for block in content_blocks: - if not isinstance(block, dict): - continue - if block.get("type") == "text": - text = block.get("text", "").strip() - if text: - last_assistant_text = text - if block.get("type") == "tool_use": - tool_name = block.get("name", "") - tool_input = block.get("input", {}) - if tool_name in ("Write", "Edit"): - fp = tool_input.get("file_path", "") - if fp: - files_modified.add(fp) - elif tool_name == "Bash": - cmd = tool_input.get("command", "") - if cmd: - bash_commands.append(cmd) - - return { - "user_messages": user_messages[-MAX_USER_MESSAGES:], - "files_modified": sorted(files_modified), - "bash_commands": bash_commands[-MAX_BASH_COMMANDS:], - "last_assistant_text": last_assistant_text[:MAX_ASSISTANT_TEXT], - } - - -def build_content(state: dict, source: str) -> str: - """Build minimal context — only what's needed to resume work. - - This is a FALLBACK safety net, not the primary capture path. - The agent handles rich memory storage via on_pre_compact.sh prompts. - This script only fires when the agent didn't store enough on its own. - - Keep it short — mem0 infer=True will extract structured facts. - """ - parts = [] - - if state["files_modified"]: - parts.append(f"Files touched: {', '.join(state['files_modified'][:15])}") - - if state["bash_commands"]: - git_cmds = [c for c in state["bash_commands"] if "git " in c] - if git_cmds: - parts.append(f"Git operations: {len(git_cmds)}") - - return "\n".join(parts) - - -def store_memory(api_key: str, content: str, user_id: str, source: str, session_id: str = "", project_id: str = "", branch: str = "") -> bool: - """Store session state as a memory via the Mem0 REST API.""" - expires = (date.today() + timedelta(days=SESSION_STATE_EXPIRY_DAYS)).isoformat() - metadata = { - "type": "session_state", - "source": source, - "session_id": session_id, - } - if branch: - metadata["branch"] = branch - body = { - "messages": [ - {"role": "user", "content": content} - ], - "user_id": user_id, - "app_id": project_id, - "metadata": metadata, - "expiration_date": expires, - "infer": True, - } - - data = json.dumps(body).encode("utf-8") - req = urllib.request.Request( - f"{API_URL}/v3/memories/add/", - data=data, - headers={ - "Content-Type": "application/json", - "Authorization": f"Token {api_key}", - }, - method="POST", - ) - - try: - with urllib.request.urlopen(req, timeout=15) as resp: - if resp.status in (200, 201): - log.info("Session state stored successfully") - return True - log.warning("API returned status %d", resp.status) - return False - except urllib.error.URLError as e: - log.warning("API call failed: %s", e) - return False - - -def format_status(state: dict, source: str, stored: bool, skipped_reason: str = "") -> str: - """Build a clean, readable status line for terminal display.""" - files_count = len(state.get("files_modified", [])) - git_cmds = [c for c in state.get("bash_commands", []) if "git " in c] - user_msgs = len(state.get("user_messages", [])) - - parts = [] - if files_count: - parts.append(f"{files_count} file{'s' if files_count != 1 else ''} touched") - if git_cmds: - parts.append(f"{len(git_cmds)} git op{'s' if len(git_cmds) != 1 else ''}") - if user_msgs: - parts.append(f"{user_msgs} exchange{'s' if user_msgs != 1 else ''}") - - activity = ", ".join(parts) if parts else "minimal activity" - - if source == "pre-compaction": - icon = "✨" # ✨ - label = "Pre-compaction snapshot" - else: - icon = "\U0001f4be" # 💾 - label = "Session-end snapshot" - - if skipped_reason: - return f"{icon} Mem0 {label} — {activity} — {skipped_reason}" - elif stored: - return f"{icon} Mem0 {label} — {activity} — saved to mem0" - else: - return f"{icon} Mem0 {label} — {activity} — nothing to capture" - - -def main(): - source = "pre-compaction" - show_status = False - for arg in sys.argv[1:]: - if arg.startswith("--source="): - source = arg.split("=", 1)[1] - elif arg == "--status": - show_status = True - - api_key = resolve_api_key() - if not api_key: - log.debug("MEM0_API_KEY not set, skipping capture") - if show_status: - print("✨ Mem0 — no API key, skipping capture") - return - - try: - hook_input = json.loads(sys.stdin.read()) - except (json.JSONDecodeError, OSError): - log.debug("No valid JSON on stdin") - return - - transcript_path = hook_input.get("transcript_path", "") - if not transcript_path: - log.debug("No transcript_path provided") - return - - session_id = hook_input.get("session_id", "") - cwd = hook_input.get("cwd") or None - user_id = resolve_user_id() - project_id = resolve_project_id(cwd) - branch = resolve_branch(cwd) - - lines = tail_lines(transcript_path, MAX_TAIL_LINES) - if not lines: - log.debug("Transcript empty or unreadable: %s", transcript_path) - return - - state = parse_transcript(lines) - - # Skip if agent already stored memories this session — avoid duplicate writes. - stats_file = f"/tmp/mem0_session_stats_{os.environ.get('USER', 'default')}.json" - try: - with open(stats_file) as f: - stats = json.load(f) - if stats.get("adds", 0) >= 1: - log.info("Agent stored %d memories this session — skipping fallback", stats["adds"]) - if show_status: - print(format_status(state, source, False, f"agent already stored {stats['adds']} memor{'ies' if stats['adds'] != 1 else 'y'}")) - return - except (OSError, json.JSONDecodeError): - pass - - if not state["files_modified"]: - log.debug("No files modified — skipping fallback capture") - if show_status: - print(format_status(state, source, False)) - return - - content = build_content(state, source) - if not content.strip(): - log.debug("No content to store") - if show_status: - print(format_status(state, source, False)) - return - - log.info("Fallback capture: %d files modified", len(state["files_modified"])) - stored = store_memory(api_key, content, user_id, source, session_id, project_id, branch) - - if show_status: - print(format_status(state, source, stored)) - - -if __name__ == "__main__": - try: - main() - except Exception as e: - log.error("Unexpected error: %s", e) - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/on_pre_compact.sh b/integrations/mem0-plugin/scripts/on_pre_compact.sh deleted file mode 100755 index fb27acb75..000000000 --- a/integrations/mem0-plugin/scripts/on_pre_compact.sh +++ /dev/null @@ -1,23 +0,0 @@ -#!/usr/bin/env bash -# Hook: PreCompact -# -# Fires BEFORE context compaction. Captures session state via REST API -# in the background. Runs silently — no output to user. - -set -uo pipefail - -if [ -n "${MEM0_DEBUG:-}" ]; then - mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" -fi - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -INPUT=$(cat) - -python3 "$SCRIPT_DIR/telemetry.py" pre_compact 2>/dev/null & - -# Capture in background, no stdout -_TMP="/tmp/mem0_precompact_input_$$.json" -printf '%s' "$INPUT" > "$_TMP" 2>/dev/null -(python3 "$SCRIPT_DIR/on_pre_compact.py" --source=pre-compaction < "$_TMP" 2>/dev/null; rm -f "$_TMP") & - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_pre_compact_cursor.sh b/integrations/mem0-plugin/scripts/on_pre_compact_cursor.sh deleted file mode 100755 index ca5a5a8e2..000000000 --- a/integrations/mem0-plugin/scripts/on_pre_compact_cursor.sh +++ /dev/null @@ -1,22 +0,0 @@ -#!/usr/bin/env bash -# Hook: preCompact (Cursor) -# -# Wraps on_pre_compact.sh and converts plain-text output to Cursor's -# expected JSON format: {"user_message":""} - -set -uo pipefail - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" - -# Pin platform so the shared script's telemetry is attributed to cursor. -export MEM0_PLATFORM=cursor - -TEXT=$("$SCRIPT_DIR/on_pre_compact.sh" 2>/dev/null || echo "") - -if [ -z "$TEXT" ]; then - echo '{}' - exit 0 -fi - -jq -cn --arg msg "$TEXT" '{user_message:$msg}' -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_session_start.sh b/integrations/mem0-plugin/scripts/on_session_start.sh deleted file mode 100755 index b2e83926b..000000000 --- a/integrations/mem0-plugin/scripts/on_session_start.sh +++ /dev/null @@ -1,207 +0,0 @@ -#!/usr/bin/env bash -set -uo pipefail - -if [ -n "${MEM0_DEBUG:-}" ]; then - mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" -fi - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -. "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true - -INPUT=$(cat) -SOURCE=$(echo "$INPUT" | jq -r '.source // "startup"' 2>/dev/null || echo "startup") - -if [ "$SOURCE" = "startup" ]; then - python3 "$SCRIPT_DIR/session_stats.py" init 2>/dev/null || true - rm -f /tmp/mem0_recent_reads_${USER:-default}_* 2>/dev/null || true -fi -PYTHONPATH="$SCRIPT_DIR" python3 "$SCRIPT_DIR/load_settings.py" init 2>/dev/null || true -rm -f "/tmp/mem0_rubric_injected_${USER:-default}" 2>/dev/null || true -rm -f /tmp/mem0_rubric_* 2>/dev/null || true -rm -f "/tmp/mem0_msg_count_${USER:-default}" 2>/dev/null || true -MEM0_SESSION_ID=$(echo "$INPUT" | jq -r '.session_id // ""' 2>/dev/null || echo "") -if [ -z "$MEM0_SESSION_ID" ]; then - MEM0_SESSION_ID="ses_$(date +%s)_$$" -fi -printf '%s' "$MEM0_SESSION_ID" > "/tmp/mem0_session_id_${USER:-default}" -export MEM0_SESSION_ID - -# Persist identity to Claude's env so Bash tool calls, MCP config, and other hooks see them -if [ -n "${CLAUDE_ENV_FILE:-}" ]; then - echo "export MEM0_SESSION_ID=\"$MEM0_SESSION_ID\"" >> "$CLAUDE_ENV_FILE" - echo "export MEM0_RESOLVED_USER_ID=\"${MEM0_RESOLVED_USER_ID:-${USER:-default}}\"" >> "$CLAUDE_ENV_FILE" - echo "export MEM0_PROJECT_ID=\"${MEM0_PROJECT_ID:-unknown}\"" >> "$CLAUDE_ENV_FILE" - echo "export MEM0_BRANCH=\"${MEM0_BRANCH:-unknown}\"" >> "$CLAUDE_ENV_FILE" - if [ -n "${MEM0_API_KEY:-}" ]; then - echo "export MEM0_API_KEY=\"$MEM0_API_KEY\"" >> "$CLAUDE_ENV_FILE" - fi -fi - -if [ -z "${MEM0_API_KEY:-}" ]; then - _UID="${MEM0_RESOLVED_USER_ID:-${USER:-default}}" - _PID="${MEM0_PROJECT_ID:-unknown}" - _BR="${MEM0_BRANCH:-unknown}" - cat </dev/null 2>&1; then - MEM0_COUNT=$(python3 -c " -import json, os, urllib.request, urllib.error -api_key = os.environ.get('MEM0_API_KEY', '') -user_id = os.environ.get('MEM0_RESOLVED_USER_ID', 'default') -app_id = os.environ.get('MEM0_PROJECT_ID', '') -global_search = os.environ.get('MEM0_GLOBAL_SEARCH', 'false') == 'true' - -def get_count(filters): - body = json.dumps({'filters': filters}).encode() - req = urllib.request.Request( - 'https://api.mem0.ai/v3/memories/?page=1&page_size=1', - headers={'Authorization': f'Token {api_key}', 'Content-Type': 'application/json'}, - data=body, method='POST', - ) - with urllib.request.urlopen(req, timeout=5) as r: - data = json.loads(r.read()) - if isinstance(data, dict) and 'count' in data: - return data['count'] - if isinstance(data, list): - return len(data) - return 0 - -try: - if global_search: - filters = {'OR': [{'user_id': '*'}]} - else: - filters = {'AND': [{'user_id': user_id}, {'app_id': app_id}]} - total = get_count(filters) - print(total) -except Exception: - print('?') -" 2>/dev/null || echo "?") -fi - -_UID="${MEM0_RESOLVED_USER_ID:-${USER:-default}}" -_ANN="${_MEM0_IDENTITY_ANNOTATION:-}" -_PID="${MEM0_PROJECT_ID:-unknown}" -_BR="${MEM0_BRANCH:-unknown}" -_GS="${MEM0_GLOBAL_SEARCH:-false}" - -if [ "$_GS" = "true" ]; then - _SCOPE_LABEL="scope=global" - _SCOPE_INSTR="Global search is ON — searches return all memories across all users and projects. Writes still use user_id: \`${_UID}\`, app_id: \`${_PID}\`." -else - _SCOPE_LABEL="project=${_PID}" - _SCOPE_INSTR="Always include \`user_id\` + \`app_id\` in every \`search_memories\` filter and \`add_memory\` call: -- user_id: \`${_UID}\` -- app_id: \`${_PID}\` (project scope — passed as top-level \`app_id\`, NOT in metadata)" -fi - -cat </dev/null || echo ".") -if command -v python3 >/dev/null 2>&1; then - MEM0_PROJECT_CONFIG=$(python3 "$SCRIPT_DIR/parse_mem0_config.py" --full "$MEM0_CWD_RESOLVED" 2>/dev/null || echo "{}") - if [ -n "$MEM0_PROJECT_CONFIG" ] && [ "$MEM0_PROJECT_CONFIG" != "{}" ]; then - _CONFIG_KEYS=$(echo "$MEM0_PROJECT_CONFIG" | python3 -c "import sys,json; d=json.load(sys.stdin); print(len(d))" 2>/dev/null || echo "?") - echo "mem0.md loaded (${_CONFIG_KEYS} sections configured)." - # Surface the project's memory policy so the model applies it when choosing - # what to store via add_memory (the hook writes also send it as an - # extraction param — see _instructions.py). - _INSTR=$(echo "$MEM0_PROJECT_CONFIG" | python3 -c "import sys,json; print(json.load(sys.stdin).get('instructions',''))" 2>/dev/null || echo "") - if [ -n "$_INSTR" ]; then - echo "" - echo "Project memory policy (from mem0.md): ${_INSTR}" - fi - fi -fi - -if [ "$SOURCE" = "startup" ]; then - if [ "$MEM0_COUNT" = "0" ]; then - echo "New project with 0 memories. Invoke the mem0:onboard skill to import project files. Coding categories install automatically in the background." - else - echo "Search mem0 for recent decisions and task learnings before responding. Run 2 parallel searches: one for decision type, one for task_learning type." - - # Inject compact recent activity timeline (non-blocking, 5s timeout) - # Use perl alarm as portable timeout (macOS lacks GNU timeout) - _TIMELINE=$(MEM0_CWD="$MEM0_CWD_RESOLVED" perl -e 'alarm 5; exec @ARGV' python3 "$SCRIPT_DIR/session_timeline.py" 2>/dev/null || echo "") - if [ -n "$_TIMELINE" ]; then - echo "" - echo "$_TIMELINE" - fi - fi - - _PROJ_KEY=$(printf '%s' "$MEM0_CWD_RESOLVED" | tr '/' '-') - _MEMORY_MD="$HOME/.claude/projects/${_PROJ_KEY}/memory/MEMORY.md" - if [ -f "$_MEMORY_MD" ] && [ -s "$_MEMORY_MD" ]; then - echo "Native MEMORY.md detected at ${_MEMORY_MD}. Add autoMemoryEnabled:false to settings.json or run /mem0:import." - fi - - MEM0_CWD="$MEM0_CWD_RESOLVED" \ - python3 "$SCRIPT_DIR/auto_import.py" 2>/dev/null & - - # Configure the coding-category taxonomy in the background (idempotent, never blocks). - # Prefer the venv python since this path needs the mem0ai SDK. - _VENV_PY="${CLAUDE_PLUGIN_DATA:-$HOME/.mem0/plugin-data}/venv/bin/python3" - if [ -x "$_VENV_PY" ]; then - MEM0_CWD="$MEM0_CWD_RESOLVED" "$_VENV_PY" "$SCRIPT_DIR/auto_setup_categories.py" 2>/dev/null & - else - MEM0_CWD="$MEM0_CWD_RESOLVED" python3 "$SCRIPT_DIR/auto_setup_categories.py" 2>/dev/null & - fi - -elif [ "$SOURCE" = "resume" ]; then - echo "Session resumed. Search mem0 for session_state and decision memories to pick up where you left off. Run 2 parallel searches." - -elif [ "$SOURCE" = "compact" ]; then - echo "Context compacted. Search mem0 for session_state and decision memories to recover context. Run 2 parallel searches." - if [ "${MEM0_AUTO_SAVE:-true}" != "false" ]; then - printf '%s' "$INPUT" | python3 "$SCRIPT_DIR/capture_compact_summary.py" 2>/dev/null & - fi -fi - -python3 "$SCRIPT_DIR/telemetry.py" session_start --source="$SOURCE" --memory_count="${MEM0_COUNT:-0}" 2>/dev/null & - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_session_start_cursor.sh b/integrations/mem0-plugin/scripts/on_session_start_cursor.sh deleted file mode 100755 index 905a5fb5b..000000000 --- a/integrations/mem0-plugin/scripts/on_session_start_cursor.sh +++ /dev/null @@ -1,22 +0,0 @@ -#!/usr/bin/env bash -# Hook: sessionStart (Cursor) -# -# Wraps on_session_start.sh and converts plain-text output to Cursor's -# expected JSON format: {"additional_context":""} - -set -uo pipefail - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" - -# Pin platform so the shared script's telemetry is attributed to cursor. -export MEM0_PLATFORM=cursor - -TEXT=$("$SCRIPT_DIR/on_session_start.sh" 2>/dev/null || echo "") - -if [ -z "$TEXT" ]; then - echo '{}' - exit 0 -fi - -jq -cn --arg ctx "$TEXT" '{additional_context:$ctx}' -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_stop.sh b/integrations/mem0-plugin/scripts/on_stop.sh deleted file mode 100755 index 636270816..000000000 --- a/integrations/mem0-plugin/scripts/on_stop.sh +++ /dev/null @@ -1,55 +0,0 @@ -#!/usr/bin/env bash -# Hook: Stop -# -# Captures a structured session summary when a Claude Code session ends. -# Parses the transcript, extracts the last assistant message and files -# touched, then stores via mem0 API with infer=True for AI extraction. -# -# Guards: -# - Skips subagent sessions (agent_id present) -# - Skips if no API key -# - Skips if no transcript_path -# - Dedup via marker file -# -# Input: JSON on stdin with transcript_path, session_id, agent_id, cwd -# Output: Nothing to stdout (background capture). Always exits 0. - -set -uo pipefail - -if [ -n "${MEM0_DEBUG:-}" ]; then - mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" -fi - -INPUT=$(cat) - -# Guard: skip subagent sessions -AGENT_ID=$(echo "$INPUT" | jq -r '.agent_id // ""' 2>/dev/null || echo "") -if [ -n "$AGENT_ID" ]; then - exit 0 -fi - -# Resolve identity if needed -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -. "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true - -# Honor auto_save=false in ~/.mem0/settings.json -if [ "${MEM0_AUTO_SAVE:-true}" = "false" ]; then - exit 0 -fi - -if [ -z "${MEM0_API_KEY:-}" ]; then - exit 0 -fi - -TRANSCRIPT_PATH=$(echo "$INPUT" | jq -r '.transcript_path // ""' 2>/dev/null || echo "") -if [ -z "$TRANSCRIPT_PATH" ]; then - exit 0 -fi - -# Run capture in the background — fires every turn now, so avoid blocking -echo "$INPUT" | python3 "$SCRIPT_DIR/capture_session_summary.py" 2>/dev/null & - -# Telemetry -python3 "$SCRIPT_DIR/telemetry.py" session_stop 2>/dev/null & - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_stop_cursor.sh b/integrations/mem0-plugin/scripts/on_stop_cursor.sh deleted file mode 100755 index 77a8cd0a4..000000000 --- a/integrations/mem0-plugin/scripts/on_stop_cursor.sh +++ /dev/null @@ -1,38 +0,0 @@ -#!/usr/bin/env bash -# Hook: stop — Cursor variant -# -# Same as on_stop.sh but uses CURSOR_PLUGIN_ROOT and Cursor-specific -# identity resolution. - -set -uo pipefail - -if [ -n "${MEM0_DEBUG:-}" ]; then - mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" -fi - -INPUT=$(cat) - -AGENT_ID=$(echo "$INPUT" | jq -r '.agent_id // ""' 2>/dev/null || echo "") -if [ -n "$AGENT_ID" ]; then - exit 0 -fi - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -# Pin platform so this hook's telemetry is attributed to cursor. -export MEM0_PLATFORM=cursor -. "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true - -if [ -z "${MEM0_API_KEY:-}" ]; then - exit 0 -fi - -TRANSCRIPT_PATH=$(echo "$INPUT" | jq -r '.transcript_path // ""' 2>/dev/null || echo "") -if [ -z "$TRANSCRIPT_PATH" ]; then - exit 0 -fi - -echo "$INPUT" | python3 "$SCRIPT_DIR/capture_session_summary.py" 2>/dev/null || true - -python3 "$SCRIPT_DIR/telemetry.py" session_stop 2>/dev/null & - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_user_prompt.sh b/integrations/mem0-plugin/scripts/on_user_prompt.sh deleted file mode 100755 index f30380003..000000000 --- a/integrations/mem0-plugin/scripts/on_user_prompt.sh +++ /dev/null @@ -1,228 +0,0 @@ -#!/usr/bin/env bash -# Hook: UserPromptSubmit -# -# Fires on every user message. Prefetches memories relevant to the current -# prompt (so relevant context is guaranteed) and also injects a decision -# rubric telling the agent when to search further itself -- including -# follow-up searches for multi-part questions. Prefetch can be disabled -# with MEM0_PREFETCH=false. -# -# Input: JSON on stdin (prompt, session_id, cwd, transcript_path) -# Output: Decision rubric injected into Claude's context (exit 0) - -# Intentionally omit -e so the script always exits 0 even if jq fails -- -# must never block the user's prompt. -set -uo pipefail - -if [ -n "${MEM0_DEBUG:-}" ]; then - mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" -fi - -INPUT=$(cat) -PROMPT=$(echo "$INPUT" | jq -r '.prompt // ""' 2>/dev/null || echo "") - -# Acknowledgements and short replies don't warrant memory context -if [ ${#PROMPT} -lt 20 ]; then - exit 0 -fi - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -# shellcheck source=_identity.sh -. "$SCRIPT_DIR/_identity.sh" 2>/dev/null || true - -# Rubric dedup: only inject full rubric once per session. -# Key on session ID to avoid cross-session interference. -SESSION_ID=$(echo "$INPUT" | jq -r '.session_id // ""' 2>/dev/null || echo "") -if [ -z "$SESSION_ID" ]; then - _SID_FILE="/tmp/mem0_session_id_${USER:-default}" - [ -f "$_SID_FILE" ] && SESSION_ID=$(cat "$_SID_FILE" 2>/dev/null) || true -fi -if [ -z "$SESSION_ID" ]; then - SESSION_ID="default_${USER:-unknown}" -fi -RUBRIC_DIR="${MEM0_RUBRIC_DIR:-/tmp}" -RUBRIC_FLAG="$RUBRIC_DIR/mem0_rubric_${SESSION_ID}" -RUBRIC_ALREADY_SHOWN="" -if [ -f "$RUBRIC_FLAG" ]; then - RUBRIC_ALREADY_SHOWN="true" -fi - -# Track message count for periodic memory-save nudges. -# Every 5th substantial message, remind the agent to store learnings. -MSG_COUNT_FILE="/tmp/mem0_msg_count_${USER:-default}" -MSG_COUNT=0 -if [ -f "$MSG_COUNT_FILE" ]; then - MSG_COUNT=$(cat "$MSG_COUNT_FILE" 2>/dev/null || echo "0") -fi -MSG_COUNT=$((MSG_COUNT + 1)) -printf '%s' "$MSG_COUNT" > "$MSG_COUNT_FILE" 2>/dev/null || true -NEEDS_SAVE_NUDGE="" -if [ $((MSG_COUNT % 5)) -eq 0 ] && [ "$MSG_COUNT" -gt 0 ]; then - NEEDS_SAVE_NUDGE="true" -fi - -# Detect stack traces and error patterns in the prompt (no API needed) -HAS_ERROR="" -if echo "$PROMPT" | grep -qE '(Traceback|panic:)'; then - HAS_ERROR="true" -elif echo "$PROMPT" | grep -qE '^\s*fatal: '; then - HAS_ERROR="true" -elif [ "$(echo "$PROMPT" | grep -cE '(Error:|Exception:|FAIL:)')" -ge 2 ]; then - HAS_ERROR="true" -fi - -# Detect file paths in the prompt (no API needed) -FILE_PATHS=$(echo "$PROMPT" | grep -oE '([a-zA-Z0-9_./-]+\.(py|ts|tsx|js|jsx|rs|go|rb|java|sh|yaml|yml|json|toml|md|sql|css|html))\b' 2>/dev/null | head -5 || echo "") - -# Detect session-resume patterns -HAS_RESUME="" -if echo "$PROMPT" | grep -qiE '(where (did )?(we|I) (leave|left) off|continue (from )?(where|last)|what were we (working|doing)|pick up where|resume (from |where)|what.s the (current|latest) (state|status)|catch me up|where are we)'; then - HAS_RESUME="true" -fi - -# Detect explicit memory-save intent -HAS_REMEMBER="" -if echo "$PROMPT" | grep -qiE '(remember (this|that)|save (this|that) (fact|info|memory|note)|store (this|that)|don.t forget (this|that)|keep (this|that) in (mind|memory))'; then - HAS_REMEMBER="true" -fi - -# Telemetry (background, fire-and-forget) -_TELEM_ARGS="" -[ -n "$HAS_ERROR" ] && _TELEM_ARGS="$_TELEM_ARGS --error_detected" -[ -n "$FILE_PATHS" ] && _TELEM_ARGS="$_TELEM_ARGS --file_paths_detected" -[ -n "$HAS_RESUME" ] && _TELEM_ARGS="$_TELEM_ARGS --resume_detected" -[ -n "$HAS_REMEMBER" ] && _TELEM_ARGS="$_TELEM_ARGS --remember_detected" -python3 "$SCRIPT_DIR/telemetry.py" user_prompt $_TELEM_ARGS 2>/dev/null & - -# No API key — emit detections only, skip search rubric -if [ -z "${MEM0_API_KEY:-}" ]; then - _PROMPT_CTX="" - if [ -n "$HAS_ERROR" ]; then - _PROMPT_CTX="Error detected in prompt. Set MEM0_API_KEY to search past debugging context." - fi - if [ -n "$FILE_PATHS" ]; then - _PROMPT_CTX="${_PROMPT_CTX:+${_PROMPT_CTX}\n}File paths detected: ${FILE_PATHS}" - fi - if [ -n "$_PROMPT_CTX" ]; then - jq -cn --arg ctx "$_PROMPT_CTX" '{ - hookSpecificOutput: { - hookEventName: "UserPromptSubmit", - additionalContext: $ctx - } - }' - fi - exit 0 -fi -USER_ID="$MEM0_RESOLVED_USER_ID" - -_PROMPT_CTX="" - -if [ -n "$HAS_RESUME" ]; then - RESUME_RESULTS=$(PYTHONPATH="$SCRIPT_DIR" MEM0_SEARCH_USER="$USER_ID" python3 -c " -import os, sys -sys.path.insert(0, os.environ.get('PYTHONPATH', '.')) -from _search import search_memories, format_results_for_context, should_rerank - -api_key = os.environ.get('MEM0_API_KEY', '') -user_id = os.environ.get('MEM0_SEARCH_USER', 'default') -project_id = os.environ.get('MEM0_PROJECT_ID', 'unknown') -rerank = should_rerank() - -state = search_memories(api_key, user_id, project_id, 'session state current task', metadata_type='session_state', top_k=3, rerank=rerank) -decisions = search_memories(api_key, user_id, project_id, 'recent decisions and learnings', metadata_type='decision', top_k=3, rerank=rerank) - -all_r = state + decisions -seen = set() -unique = [] -for m in all_r: - mid = m.get('id', '') - if mid not in seen: - seen.add(mid) - unique.append(m) - -if unique: - print(format_results_for_context(unique, heading='Session context recovered from mem0')) - print() - print('These memories provide context for resuming work.') -else: - print('No session state found in mem0.') -" 2>/dev/null || echo "") - - if [ -n "$RESUME_RESULTS" ]; then - _PROMPT_CTX="${RESUME_RESULTS}" - fi -fi - -# Query-driven prefetch: search mem0 with the current prompt and inject the top -# matches so relevant memories are guaranteed in context, not left for the agent -# to fetch. Skipped on resume (handled above with targeted queries) and when -# MEM0_PREFETCH=false. -if [ -z "$HAS_RESUME" ] && [ "${MEM0_PREFETCH:-true}" != "false" ]; then - PREFETCH_RESULTS=$(PYTHONPATH="$SCRIPT_DIR" MEM0_SEARCH_USER="$USER_ID" MEM0_SEARCH_QUERY="$PROMPT" python3 -c " -import os, sys -sys.path.insert(0, os.environ.get('PYTHONPATH', '.')) -from _search import search_memories, format_results_for_context, should_rerank - -api_key = os.environ.get('MEM0_API_KEY', '') -user_id = os.environ.get('MEM0_SEARCH_USER', 'default') -project_id = os.environ.get('MEM0_PROJECT_ID', 'unknown') -query = os.environ.get('MEM0_SEARCH_QUERY', '') - -results = search_memories(api_key, user_id, project_id, query, top_k=5, rerank=should_rerank()) -if results: - print(format_results_for_context(results, heading='Relevant memories (auto-retrieved for this request)')) -" 2>/dev/null || echo "") - - if [ -n "$PREFETCH_RESULTS" ]; then - _PROMPT_CTX="${_PROMPT_CTX:+${_PROMPT_CTX}\n}${PREFETCH_RESULTS}" - fi -fi - -if [ -n "$HAS_REMEMBER" ]; then - _PROMPT_CTX="${_PROMPT_CTX:+${_PROMPT_CTX}\n}Remember intent detected. The /mem0:remember skill auto-classifies, sets confidence=1.0, and stores verbatim." -fi - -if [ -z "$RUBRIC_ALREADY_SHOWN" ]; then - _PROMPT_CTX="${_PROMPT_CTX:+${_PROMPT_CTX}\n}Mem0 searches apply when user references past work, decision questions, errors, or non-trivial tasks. Queries use noun-phrases, 2-4 parallel calls with different metadata.type filters, and include user_id + app_id. For multi-part or comparative questions, run follow-up searches and combine results before answering -- one search is rarely enough." - touch "$RUBRIC_FLAG" 2>/dev/null || true -fi - -if [ -n "$HAS_ERROR" ]; then - _PROMPT_CTX="${_PROMPT_CTX:+${_PROMPT_CTX}\n}Error detected in prompt. Prior occurrences are available in mem0 via anti_pattern and task_learning type filters." -fi - -if [ -n "$FILE_PATHS" ]; then - _PROMPT_CTX="${_PROMPT_CTX:+${_PROMPT_CTX}\n}File paths detected: ${FILE_PATHS}" -fi - -# Auto-capture: directly call mem0 API in background every 3rd message. -# At MSG_COUNT=3 the 3rd response isn't in the transcript yet (hook fires -# before Claude responds), so we capture 4 exchanges instead of 3. The -# overlapping window ensures the next batch (MSG_COUNT=6) picks up the -# exchange that was incomplete in the previous batch. -TRANSCRIPT_PATH=$(echo "$INPUT" | jq -r '.transcript_path // ""' 2>/dev/null || echo "") -if [ "${MEM0_AUTO_SAVE:-true}" != "false" ] && [ $((MSG_COUNT % 3)) -eq 0 ] && [ "$MSG_COUNT" -gt 0 ] && [ -n "$TRANSCRIPT_PATH" ]; then - python3 "$SCRIPT_DIR/auto_capture.py" "$TRANSCRIPT_PATH" 2>/dev/null & -fi - -# Prompt-based nudge as fallback when auto-capture hasn't run yet. -_ADDS=0 -_STATS_FILE="/tmp/mem0_session_stats_${USER:-default}.json" -if [ -f "$_STATS_FILE" ]; then - _ADDS=$(python3 -c "import json; print(json.load(open('$_STATS_FILE')).get('adds',0))" 2>/dev/null || echo "0") -fi - -if [ "$MSG_COUNT" -ge 3 ] && [ "$_ADDS" -lt "$((MSG_COUNT / 3))" ]; then - _PROMPT_CTX="${_PROMPT_CTX:+${_PROMPT_CTX}\n}After responding, store any new decisions, learnings, or preferences from this exchange via add_memory. Keep it to 1 sentence per memory." -fi - -if [ -n "$_PROMPT_CTX" ]; then - jq -cn --arg ctx "$_PROMPT_CTX" '{ - hookSpecificOutput: { - hookEventName: "UserPromptSubmit", - additionalContext: $ctx - } - }' -fi - -exit 0 diff --git a/integrations/mem0-plugin/scripts/on_user_prompt_cursor.sh b/integrations/mem0-plugin/scripts/on_user_prompt_cursor.sh deleted file mode 100755 index 17a82a7f2..000000000 --- a/integrations/mem0-plugin/scripts/on_user_prompt_cursor.sh +++ /dev/null @@ -1,22 +0,0 @@ -#!/usr/bin/env bash -# Hook: beforeSubmitPrompt (Cursor) -# -# Wraps on_user_prompt.sh and converts plain-text output to Cursor's -# expected JSON format: {"continue":true,"user_message":""} - -set -uo pipefail - -SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" - -# Pin platform so the shared script's telemetry is attributed to cursor. -export MEM0_PLATFORM=cursor - -TEXT=$("$SCRIPT_DIR/on_user_prompt.sh" 2>/dev/null || echo "") - -if [ -z "$TEXT" ]; then - jq -cn '{continue:true}' - exit 0 -fi - -jq -cn --arg msg "$TEXT" '{continue:true, user_message:$msg}' -exit 0 diff --git a/integrations/mem0-plugin/scripts/parse_export_file.py b/integrations/mem0-plugin/scripts/parse_export_file.py deleted file mode 100644 index eb28ab919..000000000 --- a/integrations/mem0-plugin/scripts/parse_export_file.py +++ /dev/null @@ -1,150 +0,0 @@ -#!/usr/bin/env python3 -"""Parse a mem0 export file and output JSON. - -Input: path to a mem0-export-*.md file (sys.argv[1]) -Output: JSON array of memory records to stdout -Exit: 0 always - -Each block in the file is delimited by lines containing exactly "---". -Blocks have a YAML-like frontmatter section (key: value lines) followed -by a blank line and the memory content text. - -Example block format: ---- -id: abc123 -created_at: 2024-01-01T00:00:00Z -type: task_learnings -confidence: 0.9 -branch: main -files: src/foo.py, src/bar.py -categories: coding_conventions, task_learnings ---- -The actual memory content text goes here. - -""" - -from __future__ import annotations - -import json -import re -import sys - - -def parse_blocks(content: str) -> list[dict]: - """Split content on '---' boundaries and parse each block. - - Returns a list of dicts with keys: - id, type, confidence, branch, files (list), categories (list), content (str) - - Blocks with empty content are skipped. - Missing optional fields default to "" (scalar) or [] (list fields). - """ - # Normalise line endings - content = content.replace("\r\n", "\n").replace("\r", "\n") - - # Split on lines that are exactly "---" - raw_blocks = re.split(r"(?m)^---\s*$", content) - - # After splitting on "---", the structure for each memory is: - # raw_blocks[0] = preamble (before first ---, typically empty) - # raw_blocks[1] = frontmatter for block 1 - # raw_blocks[2] = content for block 1 - # raw_blocks[3] = frontmatter for block 2 - # raw_blocks[4] = content for block 2 - # ... - # So frontmatter blocks are at odd indices (1, 3, 5, ...) and - # content blocks at even indices (2, 4, 6, ...). - - results: list[dict] = [] - - # Pair up frontmatter + content starting at index 1 - i = 1 - while i < len(raw_blocks): - frontmatter_raw = raw_blocks[i] - content_raw = raw_blocks[i + 1] if i + 1 < len(raw_blocks) else "" - - # Parse the frontmatter key-value pairs - fm = _parse_frontmatter(frontmatter_raw) - - # Strip leading/trailing whitespace from content - memory_content = content_raw.strip() - - # Skip blocks with empty content - if not memory_content: - i += 2 - continue - - record = { - "id": fm.get("id", ""), - "type": fm.get("type", ""), - "confidence": fm.get("confidence", ""), - "branch": fm.get("branch", ""), - "files": _parse_list_field(fm.get("files", "")), - "categories": _parse_list_field(fm.get("categories", "")), - "content": memory_content, - } - - # Include created_at if present - if "created_at" in fm: - record["created_at"] = fm["created_at"] - - results.append(record) - i += 2 - - return results - - -def _parse_frontmatter(text: str) -> dict[str, str]: - """Parse simple 'key: value' lines from frontmatter text. - - Only the first colon is used as the delimiter — values may contain colons. - Lines not matching 'key: value' are ignored. - """ - result: dict[str, str] = {} - for line in text.splitlines(): - line = line.strip() - if not line: - continue - match = re.match(r"^([A-Za-z_][A-Za-z0-9_]*)\s*:\s*(.*)$", line) - if match: - key = match.group(1).strip() - value = match.group(2).strip() - result[key] = value - return result - - -def _parse_list_field(value: str) -> list[str]: - """Split a comma-separated value into a list, stripping whitespace. - - Returns [] for empty/whitespace-only input. - """ - if not value or not value.strip(): - return [] - return [item.strip() for item in value.split(",") if item.strip()] - - -def main() -> None: - if len(sys.argv) < 2: - print("Usage: parse_export_file.py ", file=sys.stderr) - print("[]") - sys.exit(0) - - filepath = sys.argv[1] - try: - if filepath == "-": - content = sys.stdin.read() - else: - with open(filepath, encoding="utf-8", errors="replace") as f: - content = f.read() - except OSError as e: - print(f"Error reading file: {e}", file=sys.stderr) - print("[]") - sys.exit(0) - - records = parse_blocks(content) - print(json.dumps(records, ensure_ascii=False, indent=2)) - sys.exit(0) - - -if __name__ == "__main__": - main() diff --git a/integrations/mem0-plugin/scripts/parse_mem0_config.py b/integrations/mem0-plugin/scripts/parse_mem0_config.py deleted file mode 100644 index dd6361f3a..000000000 --- a/integrations/mem0-plugin/scripts/parse_mem0_config.py +++ /dev/null @@ -1,316 +0,0 @@ -#!/usr/bin/env python3 -"""Parse mem0.md project configuration file. - -Reads the optional ``mem0.md`` file in a project directory and extracts -retention policies from a ``## Retention`` section. - -Retention format (inside the section): - : d — keep for N days - : forever — never prune (returned as None) - -Usage (CLI): - python3 parse_mem0_config.py [] - -Prints a JSON object mapping category names to day counts (int) or null -(forever) on stdout. Prints ``{}`` when no mem0.md or no ## Retention -section is found. -""" - -from __future__ import annotations - -import json -import os -import re -import sys - - -def find_mem0_config(cwd: str) -> str | None: - """Look for ``mem0.md`` in *cwd*. - - Returns the absolute path to ``mem0.md`` if found, else ``None``. - """ - candidate = os.path.join(cwd, "mem0.md") - return candidate if os.path.isfile(candidate) else None - - -def parse_retention(content: str) -> dict[str, int | None]: - """Parse the ``## Retention`` section of *content*. - - Scans for a heading that matches ``## Retention`` (case-insensitive), - then reads lines until the next ``##``-level heading or end of string. - - Each non-blank, non-comment line inside the section is expected to be:: - - : d → days=N (int) - : forever → days=None - - Malformed lines are silently skipped. - - Args: - content: Full text of a mem0.md file. - - Returns: - Dict mapping category name (str) to day count (int) or ``None`` - (forever). Empty dict when no ``## Retention`` section is found. - """ - # Find the ## Retention section (allow any amount of trailing whitespace / - # extra words, but the heading must start with "## Retention"). - section_match = re.search( - r"^##\s+Retention[^\n]*\n(.*?)(?=^##\s|\Z)", - content, - flags=re.MULTILINE | re.DOTALL | re.IGNORECASE, - ) - if not section_match: - return {} - - section_text = section_match.group(1) - policies: dict[str, int | None] = {} - - for line in section_text.splitlines(): - # Strip comments and whitespace - line = re.sub(r"#.*$", "", line).strip() - if not line: - continue - - # Match ": " - line_match = re.match(r"^([^:]+):\s*(.+)$", line) - if not line_match: - continue - - category = line_match.group(1).strip() - value = line_match.group(2).strip().lower() - - if value == "forever": - policies[category] = None - else: - days_match = re.match(r"^(\d+)d$", value) - if days_match: - policies[category] = int(days_match.group(1)) - # else: malformed value — skip silently - - return policies - - -def parse_section_kv(content: str, heading: str) -> dict[str, str]: - """Parse a key-value section from mem0.md. - - Looks for ``## `` (case-insensitive) and reads ``key: value`` - lines until the next ``##``-level heading or end of string. - """ - pattern = rf"^##\s+{re.escape(heading)}[^\n]*\n(.*?)(?=^##\s|\Z)" - match = re.search(pattern, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE) - if not match: - return {} - - result: dict[str, str] = {} - for line in match.group(1).splitlines(): - line = re.sub(r"#.*$", "", line).strip() - if not line: - continue - m = re.match(r"^([^:]+):\s*(.+)$", line) - if m: - result[m.group(1).strip()] = m.group(2).strip() - return result - - -def parse_section_list(content: str, heading: str) -> list[str]: - """Parse a list section from mem0.md. - - Looks for ``## `` and reads ``- item`` or bare lines. - """ - pattern = rf"^##\s+{re.escape(heading)}[^\n]*\n(.*?)(?=^##\s|\Z)" - match = re.search(pattern, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE) - if not match: - return [] - - items: list[str] = [] - for line in match.group(1).splitlines(): - line = re.sub(r"#.*$", "", line).strip() - line = re.sub(r"^[-*]\s+", "", line).strip() - if line: - items.append(line) - return items - - -def parse_section_text(content: str, heading: str) -> str: - """Parse a free-text prose section from mem0.md. - - Looks for ``## `` (case-insensitive) and returns the prose beneath - it (up to the next ``##`` heading) collapsed to a single line. Blank lines and - full-line ``#`` comments are dropped; inline ``#`` is preserved (prose may - reference e.g. issue ``#123``). - """ - pattern = rf"^##\s+{re.escape(heading)}[^\n]*\n(.*?)(?=^##\s|\Z)" - match = re.search(pattern, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE) - if not match: - return "" - - lines: list[str] = [] - for line in match.group(1).splitlines(): - stripped = line.strip() - if not stripped or stripped.startswith("#"): - continue - lines.append(stripped) - return " ".join(lines).strip() - - -def parse_ignore_patterns(content: str) -> list[str]: - """Parse the ``## Ignore`` section of *content*. - - Each non-blank line is a glob pattern (e.g., ``node_modules``, ``*.lock``). - Lines starting with ``#`` are comments and skipped. - """ - pattern = r"^##\s+Ignore[^\n]*\n(.*?)(?=^##\s|\Z)" - match = re.search(pattern, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE) - if not match: - return [] - - patterns: list[str] = [] - for line in match.group(1).splitlines(): - line = line.strip() - if not line or line.startswith("#"): - continue - line = re.sub(r"^[-*]\s+", "", line).strip() - if line: - patterns.append(line) - return patterns - - -def load_full_config(cwd: str | None = None) -> dict: - """Load all config sections from mem0.md. - - Returns a dict with keys: retention, search, categories, identity, - ignore, project_id. - Each is populated only if the corresponding ``##`` section exists. - """ - if cwd is None: - cwd = os.getcwd() - - config_path = find_mem0_config(cwd) - if config_path is None: - return {} - - try: - with open(config_path, encoding="utf-8") as fh: - content = fh.read() - except OSError: - return {} - - config: dict = {} - - retention = parse_retention(content) - if retention: - config["retention"] = retention - - search = parse_section_kv(content, "Search") - if search: - config["search"] = search - - categories = parse_section_list(content, "Categories") - if categories: - config["categories"] = categories - config["default_categories"] = categories - - identity = parse_section_kv(content, "Identity") - if identity: - config["identity"] = identity - if "project_id" in identity: - config["project_id"] = identity["project_id"] - - ignore = parse_ignore_patterns(content) - if ignore: - config["ignore"] = ignore - - settings = parse_section_kv(content, "Settings") - if settings: - config["settings"] = settings - - # Extraction policy: `## Instructions` -> custom_instructions (user/project - # scope), `## Agent Instructions` -> agent_custom_instructions (agent scope). - instructions = parse_section_text(content, "Instructions") - if instructions: - config["instructions"] = instructions - - agent_instructions = parse_section_text(content, "Agent Instructions") - if agent_instructions: - config["agent_instructions"] = agent_instructions - - return config - - -def load_retention_policies(cwd: str | None = None) -> dict[str, int | None]: - """Load retention policies from the mem0.md in *cwd*. - - Combines :func:`find_mem0_config` and :func:`parse_retention` into a - single convenience function. - """ - if cwd is None: - cwd = os.getcwd() - - config_path = find_mem0_config(cwd) - if config_path is None: - return {} - - try: - with open(config_path, encoding="utf-8") as fh: - content = fh.read() - except OSError: - return {} - - return parse_retention(content) - - -def main() -> int: - """CLI entry point. - - With ``--full``, prints the complete config. Without it, prints only - retention policies (backward-compatible). - - With ``--key ``, prints the scalar value at that path in the - full config (e.g. ``--key settings.commit_prompts``). Prints an empty - string when the key is absent. Exits 1 only on unexpected errors. - """ - full_mode = "--full" in sys.argv - - # Extract --key - key_path: str | None = None - raw_args = sys.argv[1:] - filtered_args: list[str] = [] - i = 0 - while i < len(raw_args): - if raw_args[i] == "--key" and i + 1 < len(raw_args): - key_path = raw_args[i + 1] - i += 2 - elif raw_args[i].startswith("--key="): - key_path = raw_args[i][len("--key="):] - i += 1 - elif raw_args[i].startswith("--"): - i += 1 # skip other flags like --full - else: - filtered_args.append(raw_args[i]) - i += 1 - - cwd = filtered_args[0] if filtered_args else os.getcwd() - - if key_path is not None: - config = load_full_config(cwd) - # Traverse dotted path - value: object = config - for part in key_path.split("."): - if not isinstance(value, dict): - value = None - break - value = value.get(part) - print(value if value is not None else "") - return 0 - - if full_mode: - config = load_full_config(cwd) - else: - config = load_retention_policies(cwd) - print(json.dumps(config)) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/integrations/mem0-plugin/scripts/session_stats.py b/integrations/mem0-plugin/scripts/session_stats.py deleted file mode 100644 index 357fe4dd6..000000000 --- a/integrations/mem0-plugin/scripts/session_stats.py +++ /dev/null @@ -1,138 +0,0 @@ -#!/usr/bin/env python3 -"""Session stats tracker for mem0 plugin. - -Tracks memory adds/searches per session. -Uses /tmp/mem0_session_stats_$USER.json (single file per user, reset on init). - -Usage: - python session_stats.py init # reset for new session - python session_stats.py add # record a memory write - python session_stats.py search # record a search - python session_stats.py report # print summary, clean up temp file -""" - -from __future__ import annotations - -import json -import os -import sys -from datetime import datetime - -STATS_FILE = f"/tmp/mem0_session_stats_{os.environ.get('USER', 'default')}.json" - - -def _load() -> dict: - if os.path.isfile(STATS_FILE): - try: - with open(STATS_FILE) as f: - return json.load(f) - except (json.JSONDecodeError, OSError): - pass - return { - "adds": 0, - "searches": 0, - "categories": [], - "category_counts": {}, - "started": datetime.now().isoformat(), - } - - -def _save(stats: dict) -> None: - with open(STATS_FILE, "w") as f: - json.dump(stats, f) - - -MAX_RECENT_IDS = 50 - - -def init() -> None: - _save({ - "adds": 0, - "searches": 0, - "categories": [], - "category_counts": {}, - "recent_ids": [], - "started": datetime.now().isoformat(), - }) - - -def record_add(category: str = "", memory_id: str = "") -> None: - stats = _load() - stats["adds"] = stats.get("adds", 0) + 1 - if category: - if category not in stats.get("categories", []): - stats.setdefault("categories", []).append(category) - counts = stats.setdefault("category_counts", {}) - counts[category] = counts.get(category, 0) + 1 - if memory_id: - recent = stats.setdefault("recent_ids", []) - recent.append({"id": memory_id, "category": category, "ts": datetime.now().isoformat()}) - if len(recent) > MAX_RECENT_IDS: - stats["recent_ids"] = recent[-MAX_RECENT_IDS:] - _save(stats) - - -def record_search() -> None: - stats = _load() - stats["searches"] = stats.get("searches", 0) + 1 - _save(stats) - - -def peek() -> str: - """Return current stats as JSON without clearing the file.""" - stats = _load() - return json.dumps(stats) - - -def report() -> str: - stats = _load() - adds = stats.get("adds", 0) - searches = stats.get("searches", 0) - categories = stats.get("categories", []) - - if adds == 0 and searches == 0: - return "" - - parts = [] - category_counts = stats.get("category_counts", {}) - if category_counts: - breakdown = ", ".join(f"{c} {n}" for c, n in sorted(category_counts.items(), key=lambda x: -x[1])) - parts.append(f"Session: wrote {adds} memories ({breakdown}), retrieved {searches}") - else: - parts.append(f"Session: wrote {adds} memories, retrieved {searches}") - if categories: - parts.append(f"Categories touched: {', '.join(categories)}") - - return ". ".join(parts) + "." - - -def main() -> int: - if len(sys.argv) < 2: - print("Usage: session_stats.py [init|add|search|report]", file=sys.stderr) - return 1 - - cmd = sys.argv[1] - if cmd == "init": - init() - elif cmd == "add": - category = sys.argv[2] if len(sys.argv) > 2 else "" - memory_id = sys.argv[3] if len(sys.argv) > 3 else "" - record_add(category, memory_id) - elif cmd == "search": - record_search() - elif cmd == "peek": - print(peek()) - elif cmd == "report": - result = report() - if result: - print(result) - else: - print("Session: no memory operations.") - else: - print(f"Unknown command: {cmd}", file=sys.stderr) - return 1 - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/integrations/mem0-plugin/scripts/session_timeline.py b/integrations/mem0-plugin/scripts/session_timeline.py deleted file mode 100644 index 86e22f7fb..000000000 --- a/integrations/mem0-plugin/scripts/session_timeline.py +++ /dev/null @@ -1,106 +0,0 @@ -#!/usr/bin/env python3 -"""Fetch recent memories and format a compact timeline for SessionStart. - -Searches mem0 cloud API for the most recent memories in the project -and formats them as a compact activity timeline injected below the -existing SessionStart banner. - -Input: env vars for identity (MEM0_API_KEY, MEM0_RESOLVED_USER_ID, etc.) -Output: Compact timeline text to stdout (empty if nothing found) -""" - -from __future__ import annotations - -import json -import os -import sys -import urllib.request - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -from _formatting import TYPE_ICONS, format_age -from _identity import resolve_api_key, resolve_user_id -from _project import resolve_project_id - -API_URL = "https://api.mem0.ai" -MAX_RECENT = 10 -MAX_SUMMARIES = 3 -FETCH_TIMEOUT = 5 - - -def fetch_recent_memories(api_key: str, user_id: str, project_id: str) -> list[dict]: - """Fetch the most recent memories for this project via GET list endpoint.""" - global_search = os.environ.get("MEM0_GLOBAL_SEARCH", "false") == "true" - - if global_search: - filters = {"OR": [{"user_id": "*"}]} - else: - filters = {"AND": [{"user_id": user_id}, {"app_id": project_id}]} - - body = json.dumps({"filters": filters}).encode() - req = urllib.request.Request( - f"{API_URL}/v3/memories/?page=1&page_size={MAX_RECENT}", - data=body, - headers={ - "Authorization": f"Token {api_key}", - "Content-Type": "application/json", - }, - method="POST", - ) - try: - with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r: - result = json.loads(r.read()) - if isinstance(result, dict) and "results" in result: - return result["results"][:MAX_RECENT] - if isinstance(result, list): - return result[:MAX_RECENT] - return [] - except Exception: - return [] - - -def format_timeline(memories: list[dict]) -> str: - """Format memories into a compact recent activity timeline.""" - if not memories: - return "" - - lines = ["### Recent Activity", ""] - - for m in memories: - mid = m.get("id", "?")[:8] - text = (m.get("memory", "") or "")[:120].replace("\n", " ").strip() - meta = m.get("metadata") or {} - cat = meta.get("type", "unknown") - icon = TYPE_ICONS.get(cat, "❓") - age = format_age(m) - age_str = f" ({age})" if age else "" - lines.append(f"- {icon} [{cat}]{age_str} {text} [mem0:{mid}]") - - lines.append("") - lines.append("Search mem0 for details on any of these, or for past decisions and task learnings relevant to the current task.") - - return "\n".join(lines) - - -def main(): - api_key = resolve_api_key() - if not api_key: - return - - user_id = resolve_user_id() - project_id = resolve_project_id(os.environ.get("MEM0_CWD")) - - memories = fetch_recent_memories(api_key, user_id, project_id) - if not memories: - return - - timeline = format_timeline(memories) - if timeline: - print(timeline, end="") - - -if __name__ == "__main__": - try: - main() - except Exception: - pass - sys.exit(0) diff --git a/integrations/mem0-plugin/scripts/setup_coding_categories.py b/integrations/mem0-plugin/scripts/setup_coding_categories.py deleted file mode 100644 index c3850adba..000000000 --- a/integrations/mem0-plugin/scripts/setup_coding_categories.py +++ /dev/null @@ -1,236 +0,0 @@ -#!/usr/bin/env python3 -"""Replace mem0's default category taxonomy with one tuned for coding workflows. - -mem0 auto-tags every memory with one or more `categories`. By default the list -is consumer-oriented (food, hobbies, music, ...), which is meaningless for code. -This script replaces the project's category list with a coding-focused one. - -Uses the mem0ai SDK (client.project.update). The SDK is installed into a -persistent venv at ${CLAUDE_PLUGIN_DATA}/venv by the ensure_deps.sh hook. - -Usage: - python setup_coding_categories.py # dry-run: show current vs proposed - python setup_coding_categories.py --apply # actually call project.update() - -Requires MEM0_API_KEY (or CLAUDE_PLUGIN_OPTION_MEM0_API_KEY). -""" - -from __future__ import annotations - -import argparse -import json -import os -import sys - -_script_dir = os.path.dirname(os.path.abspath(__file__)) -sys.path.insert(0, _script_dir) -from _identity import resolve_api_key # noqa: E402 - -_plugin_root = os.environ.get("CLAUDE_PLUGIN_ROOT", os.path.join(_script_dir, "..")) -_data_dir = os.environ.get("CLAUDE_PLUGIN_DATA", os.path.join(os.path.expanduser("~"), ".mem0", "plugin-data")) -_venv_site = os.path.join(_data_dir, "venv", "lib") -if os.path.isdir(_venv_site): - for d in sorted(os.listdir(_venv_site)): - sp = os.path.join(_venv_site, d, "site-packages") - if os.path.isdir(sp) and sp not in sys.path: - sys.path.insert(1, sp) - -CODING_CATEGORIES = [ - { - "architecture_decisions": ( - "Design choices, system structure, technology selection, trade-offs evaluated, " - "and architectural patterns adopted in the project." - ) - }, - { - "anti_patterns": ( - "Approaches that failed, debugging dead-ends, common mistakes to avoid, " - "and lessons learned from things that didn't work." - ) - }, - { - "task_learnings": ( - "Strategies and approaches that succeeded for specific tasks, including tooling " - "tricks, workflow shortcuts, and effective problem-solving patterns." - ) - }, - { - "tooling_setup": ( - "Development environment, build tools, dependencies, package managers, deploy " - "pipelines, and configuration steps for the project." - ) - }, - { - "bug_fixes": ( - "Specific bug fixes with root cause analysis, the fix applied, and how the bug " - "was diagnosed -- useful for recognising similar issues later." - ) - }, - { - "coding_conventions": ( - "Code style, naming patterns, file organisation, error-handling conventions, " - "and team agreements about how code is written in this project." - ) - }, - { - "user_preferences": ( - "User's stated preferences for tools, libraries, languages, formatting, " - "and ways of working." - ) - }, - { - "dependency_decisions": ( - "Why specific libraries, frameworks, or package versions were chosen or replaced, " - "including the alternatives considered and the reasoning behind the selection." - ) - }, - { - "performance_findings": ( - "Profiling results, bottlenecks identified, optimisations applied, and measurable " - "improvements achieved -- useful for avoiding regressions and guiding future work." - ) - }, - { - "security_constraints": ( - "Security requirements, authentication and authorisation rules, data-handling " - "constraints, compliance obligations, and known threat mitigations in effect." - ) - }, - { - "testing_patterns": ( - "Test strategies, frameworks chosen, coverage targets, fixture patterns, mocking " - "approaches, and how the test suite is structured for this project." - ) - }, - { - "data_model": ( - "Schema definitions, database column semantics, domain object relationships, " - "field constraints, and how data flows between storage and application layers." - ) - }, - { - "api_contracts": ( - "API endpoint shapes, request and response schemas, authentication requirements, " - "versioning policy, and any breaking-change commitments or deprecation timelines." - ) - }, - { - "deployment_runbook": ( - "How to build, release, deploy, and roll back the project. CI/CD pipeline steps, " - "environment-specific configuration, and on-call runbook entries." - ) - }, - { - "team_norms": ( - "Team working agreements, PR review etiquette, branching strategy, on-call " - "rotation, and other social or process conventions the team has agreed on." - ) - }, - { - "domain_glossary": ( - "Domain-specific terms, abbreviations, and acronyms with their precise meanings " - "in this project -- prevents misunderstandings across code, docs, and discussion." - ) - }, - { - "experiment_results": ( - "Results from A/B tests, feature-flag experiments, spikes, or proof-of-concept " - "work -- what was tried, what was measured, and what conclusion was reached." - ) - }, -] - - -def _print_categories(label: str, cats): - print(f"=== {label} ===") - if cats: - print(json.dumps(cats, indent=2)) - else: - print("(none / using mem0 defaults)") - print() - - -def _categories_match(current: list | None, proposed: list) -> bool: - """Compare categories by key sets, tolerating order differences and extra API fields.""" - if not current: - return False - current_keys = {k for d in current if isinstance(d, dict) for k in d} - proposed_keys = {k for d in proposed if isinstance(d, dict) for k in d} - if current_keys != proposed_keys: - return False - current_map = {k: v for d in current if isinstance(d, dict) for k, v in d.items()} - proposed_map = {k: v for d in proposed if isinstance(d, dict) for k, v in d.items()} - return all( - current_map.get(k, "").strip() == v.strip() - for k, v in proposed_map.items() - ) - - -def main() -> int: - ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) - ap.add_argument( - "--apply", - action="store_true", - help="Actually call project.update(). Without this flag, runs in dry-run mode.", - ) - args = ap.parse_args() - - api_key = resolve_api_key() - if not api_key: - print("ERROR: MEM0_API_KEY is not set. Export it or configure it via plugin userConfig.", file=sys.stderr) - return 1 - os.environ["MEM0_API_KEY"] = api_key - - try: - from mem0 import MemoryClient - except ImportError: - print( - "ERROR: mem0ai SDK not found. The plugin's ensure_deps.sh hook should\n" - "install it automatically on session start. Try restarting Claude Code,\n" - "or run manually: pip install mem0ai", - file=sys.stderr, - ) - return 1 - - try: - client = MemoryClient() - except Exception as e: - print( - f"ERROR initialising MemoryClient: {e}\n" - "Most commonly this is an invalid MEM0_API_KEY -- check the key at " - "https://app.mem0.ai/dashboard/api-keys", - file=sys.stderr, - ) - return 1 - - try: - current = client.project.get(fields=["custom_categories"]) - current_cats = current.get("custom_categories") if isinstance(current, dict) else None - except Exception as e: - print(f"ERROR fetching current categories: {e}", file=sys.stderr) - return 1 - - _print_categories("Current project categories", current_cats) - _print_categories("Proposed coding categories", CODING_CATEGORIES) - - if not args.apply: - print("Dry-run only -- no changes made. Re-run with --apply to write.") - return 0 - - if _categories_match(current_cats, CODING_CATEGORIES): - print("Categories already match -- skipping update.") - return 0 - - print("Applying coding categories...") - try: - response = client.project.update(custom_categories=CODING_CATEGORIES) - except Exception as e: - print(f"ERROR applying update: {e}", file=sys.stderr) - return 1 - - print("Done.", response if response else "") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/integrations/mem0-plugin/scripts/telemetry.py b/integrations/mem0-plugin/scripts/telemetry.py deleted file mode 100644 index 7f351e02c..000000000 --- a/integrations/mem0-plugin/scripts/telemetry.py +++ /dev/null @@ -1,174 +0,0 @@ -#!/usr/bin/env python3 -"""Lightweight fire-and-forget telemetry for the mem0 plugin. - -Sends anonymous usage events to PostHog using the same project key and -endpoint as the mem0 Python SDK and CLI. No posthog library dependency — -uses stdlib urllib directly (same pattern as cli/python telemetry_sender.py). - -CLI usage (called from hooks as a background subprocess): - python3 telemetry.py [--memory_count=N] [--categories_count=N] - [--error_detected] [--file_paths_detected] - [--source=] [--tool=] - -Opt-out: set MEM0_TELEMETRY=false (or 0/no/off) to disable all telemetry. - -Never sends: user content, memory content, API keys, raw user/project IDs. -Only sends: event type, platform, plugin version, anonymized hashes, counts. -""" - -from __future__ import annotations - -import hashlib -import json -import os -import platform -import sys -import urllib.error -import urllib.request - -# Each editor surface ships its own manifest with its own version line -# (Antigravity is on 0.1.x while Cursor/Codex are on 0.2.x), so we read the -# manifest matching the detected platform rather than a single shared one. -_PLATFORM_MANIFESTS = { - "antigravity": ("..", "plugin.json"), - "cursor": ("..", ".cursor-plugin", "plugin.json"), - "codex": ("..", ".codex-plugin", "plugin.json"), - "kimi": ("..", ".kimi-plugin", "plugin.json"), -} - - -def _load_plugin_version(platform_name: str = "") -> str: - parts = _PLATFORM_MANIFESTS.get(platform_name) - if parts is None: - return "unknown" - try: - plugin_json = os.path.join(os.path.dirname(__file__), *parts) - with open(plugin_json) as f: - return json.load(f).get("version", "unknown") - except (OSError, json.JSONDecodeError, KeyError): - return "unknown" - - -POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" -POSTHOG_HOST = "https://us.i.posthog.com/i/v0/e/" -REQUEST_TIMEOUT = 2 - -SAMPLE_RATE = 1.0 - - -def _sha256(value: str) -> str: - return hashlib.sha256(value.encode("utf-8")).hexdigest() - - -def _distinct_id() -> str: - """Stable anonymous ID: SHA-256 of API key if available, else SHA-256 of username.""" - api_key = os.environ.get("MEM0_API_KEY") or os.environ.get("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY") or "" - if api_key: - return hashlib.sha256(api_key.encode()).hexdigest()[:32] - user_id = os.environ.get("MEM0_RESOLVED_USER_ID") or os.environ.get("USER") or "unknown" - return _sha256(user_id) - - -def detect_platform() -> str: - explicit = os.environ.get("MEM0_PLATFORM") - if explicit: - return explicit - if os.environ.get("ANTIGRAVITY_PLUGIN_ROOT"): - return "antigravity" - if os.environ.get("KIMI_PLUGIN_ROOT"): - return "kimi" - if os.environ.get("PLUGIN_ROOT"): - return "codex" - if os.environ.get("CURSOR_PLUGIN_ROOT"): - return "cursor" - if os.environ.get("WINDSURF_PLUGIN_ROOT"): - return "windsurf" - return "plugin" - - -def is_enabled() -> bool: - return os.environ.get("MEM0_TELEMETRY", "true").lower() not in ("false", "0", "no", "off") - - -def build_posthog_payload(event_name: str, properties: dict | None = None) -> dict: - project_id = os.environ.get("MEM0_PROJECT_ID") or "unknown" - plat = detect_platform() - return { - "api_key": POSTHOG_API_KEY, - "distinct_id": _distinct_id(), - "event": event_name, - "properties": { - **(properties or {}), - "source": "plugin", - "platform": plat, - "plugin_version": _load_plugin_version(plat), - "project_hash": _sha256(project_id), - "os": sys.platform, - "os_version": platform.version(), - "sample_rate": SAMPLE_RATE, - "$process_person_profile": False, - "$lib": "posthog-python", - }, - } - - -def send(payload: dict) -> None: - data = json.dumps(payload).encode("utf-8") - req = urllib.request.Request( - POSTHOG_HOST, - data=data, - headers={"Content-Type": "application/json"}, - ) - try: - with urllib.request.urlopen(req, timeout=REQUEST_TIMEOUT): - pass - except Exception: - pass - - -def emit(event_type: str, properties: dict | None = None) -> None: - if not is_enabled(): - return - send(build_posthog_payload(f"plugin.{event_type}", properties)) - - -def main() -> int: - if not is_enabled(): - return 0 - if len(sys.argv) < 2: - return 1 - - event_type = sys.argv[1] - properties: dict = {} - - for arg in sys.argv[2:]: - if arg.startswith("--memory_count="): - try: - properties["memory_count"] = int(arg.split("=", 1)[1]) - except ValueError: - pass - elif arg.startswith("--categories_count="): - try: - properties["categories_count"] = int(arg.split("=", 1)[1]) - except ValueError: - pass - elif arg == "--error_detected": - properties["error_detected"] = True - elif arg == "--file_paths_detected": - properties["file_paths_detected"] = True - elif arg.startswith("--source="): - properties["source_detail"] = arg.split("=", 1)[1] - elif arg.startswith("--tool="): - properties["tool"] = arg.split("=", 1)[1] - elif arg.startswith("--files_count="): - try: - properties["files_count"] = int(arg.split("=", 1)[1]) - except ValueError: - pass - - emit(event_type, properties) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/integrations/mem0-plugin/skills/context-loader/SKILL.md b/integrations/mem0-plugin/skills/context-loader/SKILL.md deleted file mode 100644 index 25d3a339d..000000000 --- a/integrations/mem0-plugin/skills/context-loader/SKILL.md +++ /dev/null @@ -1,48 +0,0 @@ ---- -name: context-loader -description: Searches and injects relevant memories into context before starting work on a task. Use when beginning a new task, switching context, or when project history, past decisions, or coding conventions need to be loaded. ---- - -# Context Loader - -Pre-fetches relevant memories to prime context before working on a task. - -## When to use - -- Session start (invoke manually or auto-triggered by skill description matching) -- User starts work on a specific feature or file set -- Complex multi-step task begins -- User says "what do we know about X" or "context for X" - -## Steps - -1. **Extract topics** from current message/task. Identify: file paths, module names, feature areas, error patterns. - -2. **Run 2-4 parallel `search_memories` calls** with different angles: - - | Query angle | Filter | Purpose | - |---|---|---| - | Feature/module name | `{"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"type": "decision"}}]}` | Architecture decisions | - | File paths mentioned | `{"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"type": "convention"}}]}` | Coding patterns | - | Error keywords (if any) | `{"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"type": "anti_pattern"}}]}` | Known pitfalls | - | Broad project context | `{"AND": [{"user_id": ""}, {"app_id": ""}]}` | Catch-all | - -3. **Deduplicate** results by memory ID across all search responses. - -4. **Output compact context block** (max 10 memories): - -``` -context-loader: loaded memories for "" - - [decision] [mem0:] - - [convention] [mem0:] - - [anti_pattern] [mem0:] -``` - -5. If **zero results**: output nothing. Don't announce empty context. - -## Constraints - -- **Read-only** — never modify or delete memories -- **Max 10 memories** returned (most relevant only) -- **Silent on empty** — only surfaces findings if relevant context exists -- Skip memories already visible in current session context diff --git a/integrations/mem0-plugin/skills/dream/SKILL.md b/integrations/mem0-plugin/skills/dream/SKILL.md deleted file mode 100644 index c5319c565..000000000 --- a/integrations/mem0-plugin/skills/dream/SKILL.md +++ /dev/null @@ -1,232 +0,0 @@ ---- -name: dream -description: Consolidates stored memories by merging duplicates, resolving contradictions, and pruning stale entries. Use when memory count is high, search results feel noisy or repetitive, or periodic cleanup is needed to maintain memory quality. ---- - -# Mem0 Dream — Memory Consolidation - -This skill performs a memory consolidation pass: it fetches all project memories, -identifies near-duplicates, flags contradictions, and prunes stale entries based on -configured retention policies. All proposed changes are shown as a diff for user -approval before anything is modified. - -**IMPORTANT: Execute steps strictly in order (1 → 2 → 3 → 4 → 5 → 6). Each step depends on the previous one. Do NOT run steps in parallel or skip ahead.** - -## Step 1: Load Retention Policies - -Determine the active retention policy by running the parser script. Use the -appropriate `PLUGIN_ROOT` variable for the current platform (`${CLAUDE_PLUGIN_ROOT}`, -`${CODEX_PLUGIN_ROOT}`, or `${CURSOR_PLUGIN_ROOT}`): - -```bash -python3 "/scripts/parse_mem0_config.py" "" -``` - -Parse the JSON output (a dict of `category → days | null`). If the script fails -or returns `{}`, fall back to these built-in defaults: - -| `metadata.type` | Default retention | -|---|---| -| `session_state` | 90 days | -| `compact_summary` | 90 days | -| all others | no pruning | - -Store the resolved policies for use in Step 3. - ---- - -## Step 2: Fetch ALL Project Memories - -Call `get_memories` to retrieve every memory for the active project: - -```python -get_memories( - filters={"AND": [{"user_id": ""}, {"app_id": ""}]}, - page_size=200, -) -``` - -If the response indicates more pages exist, paginate until all memories are fetched. -Collect the full list before proceeding. If zero memories are found, print: - -``` -No memories found for project . Nothing to consolidate. -``` - -…and stop. - ---- - -## Step 3: Analyze — Find Issues - -Work entirely in-memory; do not modify anything yet. - -Group memories by `metadata.type` (use `"unknown"` when the field is absent). -For each group, identify the following: - -### 3a. Near-duplicate pairs (merge candidates) - -Two memories are near-duplicates when they express the same fact or decision but -phrased differently (e.g., "Use PostgreSQL for auth" and "Auth DB is PostgreSQL"). - -Heuristics — two memories are near-duplicates if **all** of these hold: -- Similarity threshold: estimated cosine similarity > 0.9 (use noun/keyword overlap as proxy — if >60% of significant nouns overlap, treat as >0.9 similarity). -- Same `metadata.type`. -- Neither memory is pinned (`metadata.pinned != true`). - -For each qualifying pair, draft a merged version that is more complete and specific -than either original. - -### 3b. Contradictions - -Two memories contradict when they assert opposing facts about the same topic -(e.g., "Deploy to ECS" vs. "Deploy to Vercel"). - -Identify the likely winner: the more recent memory with higher confidence wins. -Store both IDs and their content for user review. - -### 3c. Prune candidates - -A memory is a prune candidate when **any** of the following is true: - -1. Its `metadata.type` has a retention policy and the memory is older than the - configured number of days (compare `created_at` to today). -2. Its confidence score is below 0.3 AND it contains no information unique to - this project (no file paths, identifiers, or domain-specific nouns). - -**Always skip memories where `metadata.pinned == true`**, regardless of age or -confidence. - ---- - -## Step 4: Print Diff Report - -Print a structured diff to the terminal before making any changes. Use exactly -this format: - -``` -## dream — consolidation report - -Merges (): - [mem0:] + [mem0:] → "" - -Conflicts (): - [mem0:] vs [mem0:] — "" [A/B/skip] - -Prune (): - [mem0:] — , d old - -Proposed: merges, prunes, conflicts. Apply? [Y/n] -``` - -If there are zero items in any category, omit that section entirely. - -If there are zero total proposals (no merges, no prunes, no conflicts), print: - -``` -Dream complete. No duplicate, contradictory, or stale memories found. -``` - -…and stop. - ---- - -## Step 5: Wait for User Input and Apply - -### 5a. Contradictions - -For each `CONFLICT` pair in the report, wait for the user to type `A`, `B`, or -`skip` (case-insensitive). If they enter nothing (empty), treat as `skip`. - -Record the winner for each pair before proceeding to the final apply confirmation. - -### 5b. Final confirmation - -After all conflict resolutions are collected, prompt: - -``` -Apply? [Y/n] -``` - -If the user types `n` or `no` (case-insensitive), print `Cancelled. No changes made.` -and stop. - -If the user confirms (`Y`, `yes`, or empty / Enter), apply all changes in this order: - -#### Merges - -For each approved merge pair: -1. `delete_memory()` -2. `delete_memory()` -3. `add_memory` with: - - `text=""` - - `user_id=` - - `app_id=` (top-level, not in metadata) - - `metadata={"type": "", "branch": "", "confidence": , "source": "mem0-dream"}` - - `infer=False` - -#### Contradictions (resolved) - -For each resolved conflict where the user chose A or B: -- Delete the loser (the non-chosen memory): `delete_memory(memory_id=)` - -Contradictions where the user chose `skip` are left untouched. - -#### Prunes - -For each prune candidate: -- `delete_memory()` - ---- - -## Step 6: Print Summary - -After all changes are applied, print: - -``` -Dream complete — merged: , pruned: , conflicts resolved: , skipped: -``` - ---- - -## Auto mode - -When invoked with `--auto` (e.g., `/mem0:dream --auto`), run non-interactively: - -- **Merges**: applied automatically (no contradiction, both are compatible). -- **Prunes**: applied automatically (age/confidence-based, no ambiguity). -- **Contradictions**: skipped — they require human judgment. - -### Concurrency guard - -Before doing any work, check for a lock file at `/tmp/mem0_dream_auto.lock`: -- If the lock file exists and is less than 10 minutes old, print `[mem0-dream --auto] Another run in progress — skipping.` and stop. -- Otherwise, create the lock file (write the current timestamp). Delete it when done (in all exit paths). - -### Execution - -In auto mode: -1. Load policies and fetch memories (Steps 1–3) as normal. -2. Apply merges and prunes silently without printing the diff or prompting. -3. Print a compact summary: - ``` - [mem0-dream --auto] project= merged= pruned= conflicts_skipped= - ``` -4. If contradictions were detected but skipped, check if a `mem0-dream-auto` reminder already exists before storing one: - - Search for existing reminders: `search_memories(query="mem0-dream contradictions manual review", filters={"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"source": "mem0-dream-auto"}}]}, top_k=1)` - - If a result exists with similarity > 0.9, skip storing the reminder (one already exists). - - If no match, store the reminder: - ```python - add_memory( - text="mem0-dream detected contradiction(s) requiring manual review. Run /mem0:dream to resolve them interactively.", - user_id="", - app_id="", - metadata={"type": "task_learning", "source": "mem0-dream-auto", "branch": ""}, - infer=False, - ) - ``` - -## See also - -- `/mem0:forget` — targeted deletion of specific memories (search + confirm + delete) -- `/mem0:health --deep` — quick quality scan without applying changes diff --git a/integrations/mem0-plugin/skills/export/SKILL.md b/integrations/mem0-plugin/skills/export/SKILL.md deleted file mode 100644 index 67fb0ae57..000000000 --- a/integrations/mem0-plugin/skills/export/SKILL.md +++ /dev/null @@ -1,76 +0,0 @@ ---- -name: export -description: Exports all project memories to a portable Markdown file for backup or migration. Use when backing up memories, migrating to another project, sharing memory state with teammates, or archiving before cleanup. ---- - -# Mem0 Export - -Export all memories for the current project to a portable Markdown file. - -## Execution - -### Step 1: Resolve identity - -Determine the active identity: -- `user_id` from `MEM0_USER_ID` env var, else `$USER`, else `"default"` -- `project_id` (used as `app_id`) from `MEM0_PROJECT_ID` env var, or via the project resolver - -### Step 2: Fetch all memories - -Call `get_memories` with: -- `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}` -- `page_size=200` - -If the response is paginated (i.e. the result contains a `next` cursor or the count equals `page_size`), continue fetching pages until all memories are retrieved. - -### Step 3: Format each memory as a YAML-frontmatter block - -For each memory record, produce a block in this exact format: - -``` ---- -id: -created_at: -type: -confidence: -branch: -files: -categories: ---- - - -``` - -Notes: -- The `---` delimiters must be on their own lines with no extra whitespace. -- `files` and `categories` are written as comma-separated values on a single line. -- Leave a blank line after the content before the next `---` (for readability). -- If a field is missing or null, write an empty string (not "null"). - -### Step 4: Write the export file - -Determine the output filename: - -``` -mem0-export--.md -``` - -Where `` is today's date in UTC. - -Write all formatted blocks to this file using the Write tool (or equivalent). The file is written to the current working directory. - -### Step 5: Print summary - -``` -Exported memories to -``` - -Where `` is the total number of memory blocks written. - -## Error Handling - -- If `get_memories` returns an error or zero memories, print: - ``` - No memories found for project . Nothing exported. - ``` -- If the write fails, report the error to the user. diff --git a/integrations/mem0-plugin/skills/forget/SKILL.md b/integrations/mem0-plugin/skills/forget/SKILL.md deleted file mode 100644 index 2b32d865f..000000000 --- a/integrations/mem0-plugin/skills/forget/SKILL.md +++ /dev/null @@ -1,71 +0,0 @@ ---- -name: forget -description: Deletes memories by search query or memory ID with confirmation before removal. Use when removing outdated decisions, incorrect memories, sensitive data, or cleaning up after experiments. Also handles undo of recent additions. ---- - -# Mem0 Forget - -Delete specific memories from mem0. - -## Execution - -### Step 1: Parse input - -The user provides either: -- A search query: `/mem0:forget auth module decisions` -- A memory ID: `/mem0:forget ` - -If no argument, ask: "What should I forget? Provide a search query or memory ID." - -### Step 2: Find memories - -**If memory ID provided** (looks like a UUID or hex string): -- Call `get_memory` with the ID to verify it exists. -- Show: `Found: "" (created )` - -**If search query provided:** -- Call `search_memories` with: - - `query=` - - `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}` - - `top_k=10` -- Show numbered list: - ``` - Found memories matching "": - 1. (type: , created: ) [ID: ] - 2. ... - ``` - -### Step 3: Confirm - -Ask: "Delete which memories? Enter numbers (e.g., 1,3,5), 'all', or 'cancel'." - -For a single memory ID, ask: "Delete this memory? [y/N]" - -**Never delete without confirmation.** This is destructive. - -### Step 4: Delete - -For each confirmed memory, call `delete_memory` with the memory ID. - -### Step 5: Report - -``` -Deleted memories. -``` - -If any deletions failed, report which ones and why. - -## Undo recent writes - -If the user says "undo last N memories" or "undo last write": - -1. Read session stats to get recently written memory IDs: - ```bash - SCRIPT_DIR="${CLAUDE_PLUGIN_ROOT:-${CODEX_PLUGIN_ROOT:-${CURSOR_PLUGIN_ROOT:-}}}/scripts" - python3 "$SCRIPT_DIR/session_stats.py" peek - ``` -2. Parse the `recent_ids` array from the JSON output. Each entry has `id`, `category`, `ts`. -3. Show the last N entries (default 1) and ask for confirmation. -4. Delete confirmed entries via `delete_memory`. - -If `recent_ids` is empty, tell the user: "No recent memory IDs tracked this session. Try `/mem0:tour` to browse recent memories, or `/mem0:forget ` to find specific ones." diff --git a/integrations/mem0-plugin/skills/health/SKILL.md b/integrations/mem0-plugin/skills/health/SKILL.md deleted file mode 100644 index 9b5b51a8c..000000000 --- a/integrations/mem0-plugin/skills/health/SKILL.md +++ /dev/null @@ -1,160 +0,0 @@ ---- -name: health -description: Diagnoses mem0 connectivity, API key validity, and memory read/write functionality. Use when memory operations fail, searches return empty, add_memory errors occur, MCP connection drops, or to verify the plugin is working correctly. ---- - -# Mem0 Health Check - -Run a diagnostic check on the mem0 plugin. Useful for troubleshooting. - -## Execution - -Run ALL checks, then display a single summary. Do not stop on the first failure. - -### Check 1: API key - -```bash -_KEY="${MEM0_API_KEY:-${CLAUDE_PLUGIN_OPTION_MEM0_API_KEY:-}}" -[ -n "$_KEY" ] && echo "${_KEY:0:6}..." || echo "NOT_SET" -``` - -- If `NOT_SET`: FAIL — "No API key configured" -- If set: PASS — the command already prints only the first 6 chars - -### Check 2: Identity resolution - -Resolve identity using the plugin's own resolver scripts to match what hooks use: - -```bash -SCRIPT_DIR="${CLAUDE_PLUGIN_ROOT:-${CODEX_PLUGIN_ROOT:-${CURSOR_PLUGIN_ROOT:-}}}/scripts" -source "$SCRIPT_DIR/_identity.sh" 2>/dev/null -echo "user_id=${MEM0_RESOLVED_USER_ID:-}" -echo "project_id=${MEM0_PROJECT_ID:-}" -echo "branch=${MEM0_BRANCH:-}" -``` - -If `CLAUDE_PLUGIN_ROOT` is not available, fall back to: -- `user_id`: from `MEM0_USER_ID` or `$USER` -- `project_id`: from `MEM0_PROJECT_ID` or check `~/.mem0/project_map.json` for `$PWD` -- `branch`: from `git branch --show-current` - -PASS if all three are non-empty. WARN if any falls back to defaults. - -### Check 3: MCP server connectivity - -Call `search_memories` with: -- `query="health check"` -- `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}` -- `top_k=1` - -- If returns successfully (even empty): PASS -- If errors: FAIL — show the error message - -### Check 4: Memory write capability - -Call `add_memory` with: -- `text="Health check probe — safe to delete."` -- `user_id=` -- `app_id=` -- `metadata={"type": "health_check", "probe": true}` -- `infer=False` - -The response returns `event_id` (v3 writes are async). Call `get_event_status(event_id=)` to check processing. - -- If status is `SUCCEEDED`: PASS — extract the memory ID from the event result, then call `delete_memory` with that ID to clean up. -- If status is `PENDING` after 5 seconds: PASS (write accepted, processing delayed) -- If errors: FAIL — show the error. - -### Check 5: Session stats tracker - -Check if the session stats file exists and is readable: - -```bash -STATS_FILE="/tmp/mem0_session_stats_${USER}.json" -if [ -f "$STATS_FILE" ] && python3 -c "import json; json.load(open('$STATS_FILE'))" 2>/dev/null; then - echo "OK" -else - echo "FAIL" -fi -``` - -This file is created by the SessionStart hook and updated by PostToolUse hooks throughout the session. If it doesn't exist, the session hooks may not have fired yet — try sending a message first, then recheck. - -### Display - -``` -## mem0 health - -PASS API Key m0-dVe... -PASS Identity user=kartik, project=mem0, branch=main -PASS MCP Connection 142ms -PASS Write/Read write + delete OK -PASS Session Tracker stats file active - -All checks passed. -``` - -If any check fails, add a `## Troubleshooting` section with specific fix steps for each failure. - -## Extended mode: Memory Quality Analysis - -When invoked with `--deep` (e.g., `/mem0:health --deep`), run the standard 5 checks above **plus** a memory quality scan. - -### Quality Check 1: Duplicates - -Call `get_memories` with `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `page_size=200`. Compare all pairs within the same `metadata.type` group for high textual overlap (shared nouns/keywords > 60%). Report: - -``` -Potential duplicates: pairs - [mem0:] ≈ [mem0:] — both about "" -``` - -### Quality Check 2: Stale memories - -Flag memories where: -- `metadata.type` is `session_state` or `compact_summary` AND older than 90 days -- `metadata.confidence` < 0.3 AND older than 30 days - -``` -Stale candidates: - [mem0:] — session_state, 142d old -``` - -### Quality Check 2b: Low-confidence memories - -Flag memories where `metadata.confidence` < 0.5 (regardless of age). Report separately from stale: - -``` -Low-confidence memories: - [mem0:] — confidence=0.3, "" -``` - -### Quality Check 3: Contradictions - -Within each `metadata.type` group, flag pairs that assert opposing facts about the same topic. Use semantic judgment — look for negation patterns, conflicting tool/framework choices, or reversed decisions. - -``` -Possible contradictions: - [mem0:] vs [mem0:] — conflicting on "" -``` - -### Quality Check 4: Orphan memories - -Memories with no `metadata.type` set, or with `metadata.type` not in the 17 known coding categories. These were likely written without proper tagging. - -``` -Untagged/orphan memories: -``` - -### Quality summary - -``` -## Memory Quality - -Duplicates: · Stale: · Contradictions: · Orphans: -``` - -If all counts are 0: `Memory quality: clean.` -If any non-zero: append `Run /mem0:dream to fix.` - -To fix issues found by `--deep`, run `/mem0:dream` for automated consolidation (merges, prunes, conflict resolution). diff --git a/integrations/mem0-plugin/skills/import/SKILL.md b/integrations/mem0-plugin/skills/import/SKILL.md deleted file mode 100644 index 7e6c657fd..000000000 --- a/integrations/mem0-plugin/skills/import/SKILL.md +++ /dev/null @@ -1,177 +0,0 @@ ---- -name: import -description: Imports memories from an exported Markdown file or MEMORY.md into the current project. Use when migrating from another project, restoring from backup, importing Claude Code native MEMORY.md content, or setting up a new project with existing knowledge. ---- - -# Mem0 Import - -Import memories from a mem0 export file into the current project. - -## Execution - -### Step 1: Determine the export file to import - -If the user provided a filename as an argument to `/mem0:import `, use that file. - -Otherwise, list `.md` files in the current directory whose names contain `mem0-export`: - -```bash -ls -1 *.md 2>/dev/null | grep mem0-export || echo "No export files found" -``` - -If multiple files are found, ask the user which one to import. If none are found, print: -``` -No mem0-export files found in the current directory. -Run /mem0:export first, or provide the filename: /mem0:import -``` - -### Step 2: Parse the export file - -Determine the plugin root. Use the appropriate variable for the current platform: -- Claude Code: `${CLAUDE_PLUGIN_ROOT}` -- Codex: `${CODEX_PLUGIN_ROOT}` -- Cursor: `${CURSOR_PLUGIN_ROOT}` - -Run the parser script to extract memory records as JSON: - -```bash -python3 "/scripts/parse_export_file.py" "" -``` - -This outputs a JSON array where each element has: -- `id` — original memory ID (for reference only; a new ID will be assigned on import) -- `type` — metadata type -- `confidence` — metadata confidence value -- `branch` — metadata branch -- `files` — list of associated files -- `categories` — list of categories -- `content` — the memory text - -If the script fails or outputs `[]`, print: -``` -Failed to parse or file contains no valid memory blocks. -``` -and stop. - -### Step 3: Resolve identity - -Determine the active identity: -- `user_id` from `MEM0_USER_ID` env var, else `$USER`, else `"default"` -- `project_id` (used as `app_id`) from `MEM0_PROJECT_ID` env var, or via the project resolver - -### Step 4: Import each memory - -For each record in the parsed JSON array, call `add_memory` with: - -- `text=""` -- `user_id=` -- `app_id=` -- `metadata={` - - `"type": ""` (if non-empty) - - `"confidence": ""` (if non-empty) - - `"branch": ""` (if non-empty) - - `"files": ` (the list, if non-empty) - - `"source": "import"` - - `}` -- `infer=False` - -Notes: -- Do NOT pass the original `id` — the platform assigns a new ID. -- Skip records where `content` is empty (the parser already filters these, but be defensive). -- Continue importing even if individual records fail; track the count of successes. - -### Step 5: Print results - -``` -Imported memories into project -``` - -Where `` is the number of successfully imported memories. - -If any failed: -``` -Imported / memories into project ( failed) -``` - -## Importing from competing AI tools (`--tools`) - -When invoked with `--tools` (e.g., `/mem0:import --tools`), detect and import -from competing AI tool configuration files: - -### Supported tools - -| Tool | File/directory | -|------|---------------| -| Cursor | `.cursorrules` | -| GitHub Copilot | `.github/copilot-instructions.md` | -| Cline | `memory-bank/` (directory of `.md` files) | -| Continue | `.continue/rules.md` | - -### T1: Detect - -```bash -test -f .cursorrules && echo "cursor: .cursorrules" -test -f .github/copilot-instructions.md && echo "copilot: .github/copilot-instructions.md" -test -d memory-bank/ && echo "cline: memory-bank/" -test -f .continue/rules.md && echo "continue: .continue/rules.md" -``` - -### T2: Ask user - -List found files, ask which to import (numbers, comma-separated, or "all"). -If none found: -``` -No competing tool configuration files found. -Checked: .cursorrules, .github/copilot-instructions.md, memory-bank/, .continue/rules.md -``` - -### T3: Run import - -For each selected tool: -```bash -python3 "/scripts/import_competing_tools.py" --path -``` - -Tools: `cursorrules`, `copilot`, `cline`, `continue`. - -### T4: Report - -``` -Imported memories into (cursor: , copilot: ) -``` - -Notes: `infer=False`, tagged `metadata.source=-import`, sections <50 chars -skipped, chunks >10k chars truncated, safe to re-run (deduplication handles it). - ---- - -## Importing Claude Code's native MEMORY.md - -When invoked with a path to Claude Code's native `MEMORY.md` file (typically -`~/.claude/projects//memory/MEMORY.md`), or when `on_session_start.sh` -detects native auto-memory and the user chooses to import: - -1. Read the file. It contains newline-separated memory entries (one fact per line, - sometimes with `- ` bullet prefix). -2. Split by non-empty lines. Each line becomes one memory. -3. Skip lines shorter than 20 characters or lines that are just headers (`#`). -4. For each line, call `add_memory` with: - - `text=""` - - `user_id=` - - `app_id=` - - `metadata={"type": "task_learning", "source": "memory-md-import", "confidence": 0.8}` - - `infer=False` -5. Report: `Imported memories from MEMORY.md into project ` -6. Suggest disabling native auto-memory: - ``` - To avoid duplicate memory systems, add to ~/.claude/settings.json: - "autoMemoryEnabled": false - ``` - -This handles the cold-start gap when a user has been using Claude Code's native -memory and switches to mem0. - -## Error Handling - -- If the parser script is not found at `/scripts/parse_export_file.py`, print an error and stop. -- If `add_memory` calls fail consistently (e.g. auth error), report the issue and stop early. diff --git a/integrations/mem0-plugin/skills/list-projects/SKILL.md b/integrations/mem0-plugin/skills/list-projects/SKILL.md deleted file mode 100644 index 8cdbfee71..000000000 --- a/integrations/mem0-plugin/skills/list-projects/SKILL.md +++ /dev/null @@ -1,60 +0,0 @@ ---- -name: list-projects -description: Lists all projects with stored memories for the current user, showing memory counts and last activity dates. Use when checking which projects have memories, comparing memory distribution across repos, or finding a specific project scope. ---- - -# Mem0 List Projects - -Show all known project scopes for the current user. - -## Execution - -### Step 1: Fetch memories to discover app_ids - -There is no dedicated "list projects" API endpoint. Discover projects by fetching -the user's memories across all scopes. - -**Important:** A filter with only `user_id` triggers implicit null scoping — it -excludes memories that have a non-null `app_id`. Run two queries and merge: - -1. **Null-scoped:** `get_memories` with `filters={"AND": [{"user_id": ""}]}`, `page_size=200` - — catches memories without `app_id` -2. **App-scoped:** `get_memories` with `filters={"AND": [{"user_id": ""}, {"app_id": {"exists": true}}]}`, `page_size=200` - — catches memories with any `app_id` - -Run both calls in parallel. Merge results, deduplicate by memory `id`. - -If either response indicates more pages, paginate (up to 1000 total). - -### Step 2: Extract distinct projects - -For each memory, determine project by: -1. Top-level `app_id` field (preferred) -2. `metadata.project_id` (legacy memories) -3. `metadata.project` (oldest format) -4. `"(unscoped)"` if none found - -Group by resolved project name. For each project, count: -- Total memories -- Most recent `created_at` date -- Top 3 `metadata.type` values by frequency - -### Step 3: Display - -``` -## mem0 projects - - memories (last: ) ← current - memories (last: ) - - projects, total memories -``` - -Mark current project with `← current`. Sort by memory count descending. - -### Step 4: Empty state - -If zero memories found: -``` -No projects found. Run /mem0:onboard to get started. -``` diff --git a/integrations/mem0-plugin/skills/mem0/LICENSE b/integrations/mem0-plugin/skills/mem0/LICENSE deleted file mode 100644 index 78c99ae28..000000000 --- a/integrations/mem0-plugin/skills/mem0/LICENSE +++ /dev/null @@ -1,189 +0,0 @@ - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but not - limited to compiled object code, generated documentation, and - conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work. - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to the Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by the Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding any notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - Copyright 2024 Mem0.ai - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/integrations/mem0-plugin/skills/mem0/README.md b/integrations/mem0-plugin/skills/mem0/README.md deleted file mode 100644 index 4ed29d162..000000000 --- a/integrations/mem0-plugin/skills/mem0/README.md +++ /dev/null @@ -1,73 +0,0 @@ -# Mem0 Skill for Claude - -Add persistent memory to any AI application in minutes using [Mem0 Platform](https://app.mem0.ai?utm_source=oss&utm_medium=mem0-plugin-skill-readme). - -## What This Skill Does - -When installed, Claude can: - -- **Set up Mem0** in your Python or TypeScript project -- **Integrate memory** into your existing AI app (LangChain, CrewAI, Vercel AI, OpenAI Agents, LangGraph, LlamaIndex, etc.) -- **Generate working code** using real API references and tested patterns -- **Search live docs** on demand for the latest Mem0 documentation - -## Installation - -This skill is included automatically when you install the Mem0 plugin: - -``` -/plugin marketplace add mem0ai/mem0 -/plugin install mem0@mem0-plugins -``` - -See the [plugin README](../../README.md) for full setup instructions. - -### Prerequisites - -- A Mem0 Platform API key ([Get one here](https://app.mem0.ai/dashboard/api-keys?utm_source=oss&utm_medium=mem0-plugin-skill-readme)) -- Python 3.10+ or Node.js 18+ -- Set the environment variable: - - ```bash - export MEM0_API_KEY="m0-your-api-key" - ``` - -## Quick Start - -After installing, just ask Claude: - -- "Set up mem0 in my project" -- "Add memory to my chatbot" -- "Help me search user memories with filters" -- "Integrate mem0 with my LangChain app" -- "Add graph memory to track entity relationships" - -## What's Inside - -```text -skills/mem0/ -├── SKILL.md # Skill definition and instructions -├── README.md # This file -├── LICENSE # Apache-2.0 -├── scripts/ -│ └── mem0_doc_search.py # Search live Mem0 docs on demand -└── references/ # Documentation (loaded on demand) - ├── quickstart.md # Full quickstart (Python, TS, cURL) - ├── sdk-guide.md # All SDK methods (Python + TypeScript) - ├── api-reference.md # REST endpoints, filters, memory object - ├── architecture.md # Processing pipeline, lifecycle, scoping, performance - ├── features.md # Retrieval, graph, categories, MCP, webhooks, multimodal - ├── integration-patterns.md # LangChain, CrewAI, Vercel AI, LangGraph, LlamaIndex, etc. - └── use-cases.md # 7 real-world patterns with Python + TypeScript code -``` - -## Links - -- [Mem0 Platform Dashboard](https://app.mem0.ai?utm_source=oss&utm_medium=mem0-plugin-skill-readme) -- [Mem0 Documentation](https://docs.mem0.ai) -- [Mem0 GitHub](https://github.com/mem0ai/mem0) -- [API Reference](https://docs.mem0.ai/api-reference) - -## License - -Apache-2.0 diff --git a/integrations/mem0-plugin/skills/mem0/SKILL.md b/integrations/mem0-plugin/skills/mem0/SKILL.md deleted file mode 100644 index 24cd827f2..000000000 --- a/integrations/mem0-plugin/skills/mem0/SKILL.md +++ /dev/null @@ -1,187 +0,0 @@ ---- -name: mem0 -description: Mem0 SDK reference covering Python and TypeScript APIs, memory client methods, configuration, and framework integrations. Use when writing code that calls mem0 APIs, configuring memory providers, or integrating mem0 into an application. -license: Apache-2.0 -metadata: - author: mem0ai - version: "0.1.1" - category: ai-memory - tags: "memory, personalization, ai, python, typescript, vector-search" -compatibility: Requires Python 3.10+ or Node.js 18+, pip install mem0ai or npm install mem0ai, MEM0_API_KEY env var (Platform), and internet access to api.mem0.ai. Uses Mem0 v3 API. ---- - -# Mem0 Platform Integration - -> **Skill Graph:** This skill is part of the Mem0 skill graph: -> - **mem0** (this skill) -- Platform Client SDK + OSS (Python + TypeScript) -> - **[mem0-vercel-ai-sdk](https://github.com/mem0ai/mem0/tree/main/skills/mem0-vercel-ai-sdk)** -- Vercel AI SDK provider - -Mem0 is a managed memory layer for AI applications. It stores, retrieves, and manages user memories via API — no infrastructure to deploy. For self-hosted usage, see the OSS section in the client references below. - -## Step 1: Install and authenticate - -**Python:** -```bash -pip install mem0ai -export MEM0_API_KEY="m0-your-api-key" -``` - -**TypeScript/JavaScript:** -```bash -npm install mem0ai -export MEM0_API_KEY="m0-your-api-key" -``` - -Get an API key at: https://app.mem0.ai/dashboard/api-keys?utm_source=oss&utm_medium=mem0-plugin-skill - -> **Don't have a `MEM0_API_KEY`?** Sign up at https://app.mem0.ai and create one from the dashboard. Keys start with `m0-`. - -## Step 2: Initialize the client - -**Python:** -```python -from mem0 import MemoryClient -client = MemoryClient(api_key="m0-xxx") -``` - -**TypeScript:** -```typescript -import MemoryClient from 'mem0ai'; -const client = new MemoryClient({ apiKey: 'm0-xxx' }); -``` - -For async Python, use `AsyncMemoryClient`. - -## Step 3: Core operations - -Every Mem0 integration follows the same pattern: **retrieve → generate → store**. - -### Add memories -```python -messages = [ - {"role": "user", "content": "I'm a vegetarian and allergic to nuts."}, - {"role": "assistant", "content": "Got it! I'll remember that."} -] -client.add(messages, user_id="alice") -``` - -### Search memories -```python -results = client.search("dietary preferences", filters={"user_id": "alice"}) -for mem in results.get("results", []): - print(mem["memory"]) -``` - -### Get all memories -```python -all_memories = client.get_all(filters={"user_id": "alice"}) -``` - -### Update a memory -```python -client.update("memory-uuid", text="Updated: vegetarian, nut allergy, prefers organic") -``` - -### Delete a memory -```python -client.delete("memory-uuid") -client.delete_all(user_id="alice") # delete all for a user -``` - -## Common integration pattern - -```python -from mem0 import MemoryClient -from openai import OpenAI - -mem0 = MemoryClient() -openai = OpenAI() - -def chat(user_input: str, user_id: str) -> str: - # 1. Retrieve relevant memories - memories = mem0.search(user_input, filters={"user_id": user_id}) - context = "\n".join([m["memory"] for m in memories.get("results", [])]) - - # 2. Generate response with memory context - response = openai.chat.completions.create( - model="gpt-5-mini", - messages=[ - {"role": "system", "content": f"User context:\n{context}"}, - {"role": "user", "content": user_input}, - ] - ) - reply = response.choices[0].message.content - - # 3. Store interaction for future context - mem0.add( - [{"role": "user", "content": user_input}, {"role": "assistant", "content": reply}], - user_id=user_id - ) - return reply -``` - -## Common edge cases - -- **Search returns empty:** v3 processes `add()` asynchronously — returns an event ID immediately. Wait 2-3s before searching. Also verify `user_id` matches exactly (case-sensitive) and use `filters={"user_id": "..."}` syntax. -- **AND filter with user_id + agent_id returns empty:** Entities are stored separately. `{"AND": [{"user_id": "alice"}, {"agent_id": "bot"}]}` returns nothing. Use `OR` instead, or query each separately. -- **Duplicate memories:** Don't mix `infer=True` (default) and `infer=False` for the same data. `infer=True` extracts facts via LLM with dedup. `infer=False` stores raw — same text can be stored twice. -- **Implicit null scoping:** `filters={"user_id": "alice"}` only returns memories where `agent_id`, `app_id`, `run_id` are ALL null. Wrap in `{"OR": [...]}` to include memories with non-null scoping fields. -- **Platform vs OSS imports:** Platform: `from mem0 import MemoryClient`. OSS: `from mem0 import Memory`. Don't mix them — `MemoryClient` talks to `api.mem0.ai`, `Memory` runs locally. -- **v3 defaults:** `top_k=20`, `threshold=0.1`, `rerank=False`. Adjust as needed. - -## v3 API (Current) - -Mem0 v3 uses single-pass extraction, entity linking, and multi-signal retrieval. - -**Key v3 changes from v2:** -- **Endpoints:** `POST /v3/memories/add/`, `POST /v3/memories/search/`, `POST /v3/memories/` (paginated list) -- **Extraction:** Single ADD-only pass — no more UPDATE/DELETE operations during extraction. Memories accumulate rather than consolidate. -- **Entity linking:** Replaces graph memory. Auto-extracted during `add()`, no config needed. Remove `enable_graph` and `graph_store` from any old config. -- **Defaults:** `top_k=20`, `threshold=0.1`, `rerank=False` -- **Removed params:** `org_id`, `project_id`, `enable_graph` — all removed from SDK -- **TypeScript:** Exclusively camelCase (`userId`, `agentId`, `appId`, `topK`) -- **Add response:** Async — returns event ID immediately, poll via `GET /v1/event/{event_id}/` - -See the [migration guide](https://docs.mem0.ai/migration/platform-v2-to-v3) for details. - -## Live documentation search - -For the latest docs beyond what's in the references, use the doc search tool: - -```bash -python ${CLAUDE_SKILL_DIR}/scripts/mem0_doc_search.py --query "topic" -python ${CLAUDE_SKILL_DIR}/scripts/mem0_doc_search.py --page "/platform/features/graph-memory" -python ${CLAUDE_SKILL_DIR}/scripts/mem0_doc_search.py --index -``` - -No API key needed — searches docs.mem0.ai directly. - -## Client SDK References - -Language-specific deep references (Platform + OSS): - -| Language | File | -|----------|------| -| Python (MemoryClient + AsyncMemoryClient + Memory OSS) | [client/python.md](client/python.md) | -| TypeScript/Node.js (MemoryClient + Memory OSS) | [client/node.md](client/node.md) | -| Python vs TypeScript differences | [client/differences.md](client/differences.md) | - -## Platform References - -Load these on demand for deeper detail: - -| Topic | File | -|-------|------| -| Quickstart (Python, TS, cURL) | [references/quickstart.md](references/quickstart.md) | -| SDK guide (all methods, both languages) | [references/sdk-guide.md](references/sdk-guide.md) | -| API reference (endpoints, filters, object schema) | [references/api-reference.md](references/api-reference.md) | -| Architecture (pipeline, lifecycle, scoping, performance) | [references/architecture.md](references/architecture.md) | -| Platform features (retrieval, graph, categories, MCP, etc.) | [references/features.md](references/features.md) | -| Framework integrations (LangChain, CrewAI, OpenAI Agents, etc.) | [references/integration-patterns.md](references/integration-patterns.md) | -| Use cases & examples (real-world patterns with code) | [references/use-cases.md](references/use-cases.md) | - -## Related Mem0 Skills - -| Skill | When to use | Link | -|-------|-------------|------| -| mem0-vercel-ai-sdk | Vercel AI SDK provider with automatic memory | [GitHub](https://github.com/mem0ai/mem0/tree/main/skills/mem0-vercel-ai-sdk) | diff --git a/integrations/mem0-plugin/skills/mem0/client/differences.md b/integrations/mem0-plugin/skills/mem0/client/differences.md deleted file mode 100644 index e5e80e250..000000000 --- a/integrations/mem0-plugin/skills/mem0/client/differences.md +++ /dev/null @@ -1,129 +0,0 @@ -# Python vs TypeScript SDK Differences - -Quick-reference cheatsheet for developers working across both Mem0 SDKs. - -## Constructor - -| Aspect | Python | TypeScript | -|--------|--------|------------| -| Import (Platform) | `from mem0 import MemoryClient` | `import MemoryClient from 'mem0ai'` | -| Import (OSS) | `from mem0 import Memory` | `import { Memory } from 'mem0ai/oss'` | -| Constructor | `MemoryClient(api_key="m0-xxx")` | `new MemoryClient({ apiKey: 'm0-xxx' })` | -| Required param | `api_key` (positional or kwarg) | `apiKey` (in options object) | - -Both read from `MEM0_API_KEY` env var if no key provided. - -## Method Naming - -| Operation | Python | TypeScript | -|-----------|--------|------------| -| Add | `add()` | `add()` | -| Search | `search()` | `search()` | -| Get | `get()` | `get()` | -| Get all | `get_all()` | `getAll()` | -| Update | `update()` | `update()` | -| Delete | `delete()` | `delete()` | -| Delete all | `delete_all()` | `deleteAll()` | -| History | `history()` | `history()` | -| Batch update | `batch_update()` | `batchUpdate()` | -| Batch delete | `batch_delete()` | `batchDelete()` | -| List users | `users()` | `users()` | -| Delete users | `delete_users()` | `deleteUsers()` | -| Get project | `project.get()` | `getProject()` | -| Update project | `project.update()` | `updateProject()` | -| Create webhook | `create_webhook()` | `createWebhook()` | -| Get webhooks | `get_webhooks()` | `getWebhooks()` | -| Update webhook | `update_webhook()` | `updateWebhook()` | -| Delete webhook | `delete_webhook()` | `deleteWebhook()` | -| Create export | `create_memory_export()` | `createMemoryExport()` | -| Get export | `get_memory_export()` | `getMemoryExport()` | -| Feedback | `feedback()` | `feedback()` | - -**Rule:** Python uses `snake_case`, TypeScript uses `camelCase` for method names. - -## Parameter Passing - -```python -# Python: kwargs -client.add(messages, user_id="alice", metadata={"source": "chat"}) -client.search("query", filters={"user_id": "alice"}, top_k=5, rerank=True) -``` - -```typescript -// TypeScript: options object with camelCase for top-level params, snake_case for filter keys -await client.add(messages, { userId: 'alice', metadata: { source: 'chat' } }); -await client.search('query', { filters: { user_id: 'alice' }, topK: 5, rerank: true }); -``` - -**v3:** Python uses `snake_case` everywhere. TypeScript uses `camelCase` for top-level params (`userId`, `topK`) but `snake_case` for filter keys (`user_id`, `agent_id`). - -## Architectural Differences - -| Aspect | Python | TypeScript | -|--------|--------|------------| -| HTTP library | httpx | axios | -| Default timeout | 300s | 60s | -| Sync support | Yes (`MemoryClient`) | No (all async) | -| Async support | Yes (`AsyncMemoryClient`) | All methods are async | -| Project management | `client.project.*` (separate class) | `client.getProject()` / `client.updateProject()` | -| Context manager | `async with AsyncMemoryClient()` | Not supported | - -## Platform Features: Python-only - -These methods exist in Python but not TypeScript: - -| Method | Description | -|--------|-------------| -| `get_summary(filters)` | Get summary of memories | -| `reset()` | Delete ALL data (users + memories) | -| `project.create(name)` | Create a new project | -| `project.delete()` | Delete current project | -| `project.get_members()` | List project members | -| `project.add_member(email, role)` | Add member to project | -| `project.update_member(email, role)` | Change member role | -| `project.remove_member(email)` | Remove member | - -## Platform Features: TypeScript-only - -| Method | Description | -|--------|-------------| -| `deleteUser(data)` | Convenience method for single entity deletion | -| `ping()` | Health check endpoint | - -## OSS Config Naming - -| Python config key | TypeScript config key | -|-------------------|----------------------| -| `vector_store` | `vectorStore` | -| `history_db_path` | `historyDbPath` | -| `custom_instructions` | `customInstructions` | - -## OSS Scope Parameter Naming - -| Python | TypeScript | -|--------|------------| -| `user_id="alice"` | `userId: 'alice'` | -| `agent_id="bot"` | `agentId: 'bot'` | -| `run_id="session"` | `runId: 'session'` | - -## Entity ID Passing (v3) - -| Method | Python | TypeScript | -|--------|--------|------------| -| add() | Top-level: `user_id="alice"` | Top-level: `{ userId: 'alice' }` | -| search() | In filters: `filters={"user_id": "alice"}` | In filters: `{ filters: { user_id: 'alice' } }` | -| get_all() | In filters: `filters={"user_id": "alice"}` | In filters: `{ filters: { user_id: 'alice' } }` | - -## Common Gotcha - -When searching/filtering, both Python and TypeScript use `snake_case` for filter keys. TypeScript only uses `camelCase` for top-level method parameters: - -```python -# Python - snake_case in filters -results = client.search("query", filters={"user_id": "alice"}) -``` - -```typescript -// TypeScript - snake_case in filters, camelCase for top-level params -const results = await client.search('query', { filters: { user_id: 'alice' }, topK: 20 }); -``` diff --git a/integrations/mem0-plugin/skills/mem0/client/node.md b/integrations/mem0-plugin/skills/mem0/client/node.md deleted file mode 100644 index ca85ba7ba..000000000 --- a/integrations/mem0-plugin/skills/mem0/client/node.md +++ /dev/null @@ -1,418 +0,0 @@ -# Mem0 Node.js / TypeScript SDK Reference - -Complete reference for the `mem0ai` npm package. Covers both the Platform client (managed API) and the Open Source self-hosted variant. - ---- - -## Platform Client - -### Installation - -```bash -npm install mem0ai -export MEM0_API_KEY="m0-your-api-key" -``` - -### MemoryClient - -```typescript -import MemoryClient from 'mem0ai'; - -const client = new MemoryClient({ apiKey: 'm0-xxx' }); -``` - -**Constructor:** `new MemoryClient({ apiKey })`. If `apiKey` is not provided, reads from `MEM0_API_KEY` environment variable. - -- HTTP library: `axios` -- Timeout: 60 seconds -- Base URL: `https://api.mem0.ai` -- All methods are async (return `Promise`) - ---- - -### Memory Methods - -#### add(messages, options?) - -Store new memories from messages. - -```typescript -const messages = [ - { role: 'user', content: "I'm a vegetarian and allergic to nuts." }, - { role: 'assistant', content: "Got it! I'll remember that." }, -]; -await client.add(messages, { userId: 'alice' }); -``` - -| Parameter | Type | Description | -|-----------|------|-------------| -| `messages` | `Message[]` | Array of `{role, content}` objects | -| `options.userId` | string | User identifier | -| `options.agentId` | string | Agent identifier | -| `options.appId` | string | Application identifier | -| `options.runId` | string | Session identifier | -| `options.metadata` | object | Custom key-value pairs | -| `options.infer` | boolean | If false, store raw text (default: true) | - -**Returns:** `Promise` -- list of events - -#### search(query, options?) - -Search memories by semantic similarity. - -```typescript -const results = await client.search('dietary preferences', { filters: { user_id: 'alice' }, topK: 20 }); -for (const mem of results.results) { - console.log(mem.memory, mem.score); -} -``` - -| Parameter | Type | Description | -|-----------|------|-------------| -| `query` | string | Natural language search query | -| `options.filters` | object | Filter object with entity IDs (`user_id`, `agent_id`, etc.) and/or `AND`/`OR`/`NOT` conditions | -| `options.topK` | number | Number of results (default: 20) | -| `options.rerank` | boolean | Enable semantic reranking (default: false) | -| `options.threshold` | number | Minimum similarity (default: 0.1) | - -**Returns:** `Promise` -- `{results: [{id, memory, score, ...}]}` - -#### get(memoryId) - -```typescript -const memory = await client.get('ea925981-...'); -``` - -#### getAll(options?) - -Retrieve all memories. Requires at least one entity identifier in filters. - -```typescript -const memories = await client.getAll({ filters: { user_id: 'alice' } }); -// With filters -const filtered = await client.getAll({ - filters: { AND: [{ user_id: 'alice' }, { categories: { contains: 'health' } }] }, -}); -``` - -| Parameter | Type | Description | -|-----------|------|-------------| -| `options.filters` | object | Filter object with entity IDs (`user_id`, `agent_id`, etc.) and/or `AND`/`OR`/`NOT` conditions | -| `options.page` | number | Page number | -| `options.pageSize` | number | Results per page | - -#### update(memoryId, data) - -```typescript -await client.update('ea925981-...', { text: 'Updated: vegan since 2024' }); -await client.update('ea925981-...', { text: 'Updated', metadata: { verified: true } }); -``` - -| Parameter | Type | Description | -|-----------|------|-------------| -| `memoryId` | string | Memory ID | -| `data.text` | string | New content | -| `data.metadata` | object | New metadata | -| `data.timestamp` | string | New timestamp | - -#### delete(memoryId) - -```typescript -await client.delete('ea925981-...'); -``` - -#### deleteAll(options?) - -```typescript -await client.deleteAll({ userId: 'alice' }); -``` - -#### history(memoryId) - -```typescript -const history = await client.history('ea925981-...'); -// Returns: [{previousValue, newValue, action, timestamps}] -``` - ---- - -### Batch Methods - -#### batchUpdate(memories) - -```typescript -await client.batchUpdate([ - { memoryId: 'uuid-1', text: 'Updated text' }, - { memoryId: 'uuid-2', text: 'Another update' }, -]); -``` - -#### batchDelete(memories) - -```typescript -await client.batchDelete(['uuid-1', 'uuid-2', 'uuid-3']); -``` - ---- - -### User/Entity Management - -#### users() - -```typescript -const users = await client.users(); -// Returns: {results: [{type: "user", name: "alice"}, ...]} -``` - -#### deleteUser(data) / deleteUsers(data) - -```typescript -await client.deleteUser({ userId: 'alice' }); // Single entity -await client.deleteUsers({ agentId: 'bot-1' }); // Flexible -``` - ---- - -### Project Management - -```typescript -// Get project config -const config = await client.getProject({ fields: ['customCategories'] }); - -// Update project settings -await client.updateProject({ - customInstructions: 'Extract dietary preferences and health info', - customCategories: [{ health: 'Medical and dietary info' }], -}); -``` - ---- - -### Webhooks - -```typescript -// List -const webhooks = await client.getWebhooks({ projectId: 'proj_123' }); - -// Create -const webhook = await client.createWebhook({ - url: 'https://your-app.com/webhook', - name: 'Memory Logger', - projectId: 'proj_123', - eventTypes: ['memory_add', 'memory_update'], -}); - -// Update -await client.updateWebhook({ - webhookId: 'wh_123', - name: 'Updated Logger', - url: 'https://new-url.com', -}); - -// Delete -await client.deleteWebhook({ webhookId: 'wh_123' }); -``` - ---- - -### Feedback - -```typescript -await client.feedback({ - memoryId: 'mem-123', - feedback: 'POSITIVE', - feedbackReason: 'Accurately captured preference', -}); -``` - ---- - -### Export - -```typescript -const exportReq = await client.createMemoryExport({ - schema: JSON.stringify({ type: 'object', properties: { name: { type: 'string' } } }), - filters: { user_id: 'alice' }, -}); - -const result = await client.getMemoryExport({ memoryExportId: exportReq.id }); -``` - ---- - -### TypeScript Types - -Key interfaces from `mem0.types.ts`: - -```typescript -interface Message { role: string; content: string; } -interface Memory { id: string; memory: string; userId: string; categories: string[]; score?: number; /* ... */ } -interface MemoryOptions { userId?: string; agentId?: string; appId?: string; runId?: string; metadata?: object; /* ... */ } -interface SearchOptions { filters?: object; topK?: number; rerank?: boolean; threshold?: number; /* ... */ } -interface MemoryHistory { id: string; memoryId: string; previousValue: string; newValue: string; action: string; /* ... */ } -interface FeedbackPayload { memoryId: string; feedback: string; feedbackReason?: string; } -interface WebhookCreatePayload { url: string; name: string; projectId: string; eventTypes: string[]; } -``` - ---- - -## Open Source / Self-Hosted - -### Installation - -```bash -npm install mem0ai -``` - -### Memory Class - -```typescript -import { Memory } from 'mem0ai/oss'; - -const m = new Memory(); // Uses default config -``` - -**Import:** `from 'mem0ai/oss'` (NOT the default export -- that is `MemoryClient` for Platform) - -### Configuration - -```typescript -const config = { - llm: { - provider: 'openai', // openai, groq, anthropic, google, ollama, lmstudio, mistral, azure - config: { - model: 'gpt-5-mini', - apiKey: 'sk-xxx', - }, - }, - embedder: { - provider: 'openai', // openai, ollama, lmstudio, google, azure, langchain, anthropic - config: { - model: 'text-embedding-3-small', - apiKey: 'sk-xxx', - }, - }, - vectorStore: { - provider: 'qdrant', // memory, qdrant, redis, supabase, langchain, azure_ai_search, pgvector - config: { - collectionName: 'my_memories', - host: 'localhost', - port: 6333, - }, - }, - historyDbPath: 'history.db', - customInstructions: '...', - disableHistory: false, -}; - -const m = new Memory(config); -// Or from dict with validation: -const m2 = Memory.fromConfig(config); -``` - -### Methods - -All methods are async (return `Promise`): - -#### add(messages, config) - -```typescript -await m.add('I prefer dark mode', { userId: 'alice' }); -await m.add([ - { role: 'user', content: 'I like hiking' }, - { role: 'assistant', content: 'Great outdoor activity!' }, -], { userId: 'alice' }); -``` - -| Parameter | Type | Description | -|-----------|------|-------------| -| `messages` | `string \| Message[]` | Content to store | -| `config.userId` | string | User identifier (at least one scope required) | -| `config.agentId` | string | Agent identifier | -| `config.runId` | string | Session identifier | -| `config.metadata` | object | Custom key-value pairs | -| `config.filters` | object | Additional filters | -| `config.infer` | boolean | LLM inference (default: true) | - -**Returns:** `Promise<{results: [...], relations?: [...]}>` - -#### search(query, config) - -```typescript -const results = await m.search('dietary preferences', { filters: { user_id: 'alice' }, topK: 5 }); -``` - -| Parameter | Type | Description | -|-----------|------|-------------| -| `query` | string | Search query | -| `config.filters` | object | Filter object with entity IDs (`user_id`, `agent_id`, `run_id`, etc.) | -| `config.topK` | number | Max results (default: 20) | - -#### get(memoryId) / getAll(config) / update(memoryId, data) / delete(memoryId) / deleteAll(config) / history(memoryId) - -Same interface patterns. Note: OSS `update` takes a string for data, not an object. - -```typescript -await m.update('mem-id', 'new content'); -``` - -#### reset() - -Clear the entire vector store and history. - -```typescript -await m.reset(); -``` - ---- - -## Key Differences: Platform vs OSS - -| Aspect | Platform (`MemoryClient`) | OSS (`Memory`) | -|--------|--------------------------|----------------| -| **Import** | `import MemoryClient from 'mem0ai'` | `import { Memory } from 'mem0ai/oss'` | -| **Auth** | API key required (`MEM0_API_KEY`) | No API key -- config-based | -| **Execution** | API calls to `api.mem0.ai` | Local execution | -| **Infrastructure** | Fully managed | Self-managed vector DB, embedder, LLM | -| **Param style** | Top-level: `camelCase` (`userId`, `topK`), filter keys: `snake_case` (`user_id`) | Top-level: `camelCase` (`userId`, `topK`), filter keys: `snake_case` (`user_id`) | -| **Batch ops** | `batchUpdate`, `batchDelete` | Not available | -| **Webhooks** | Full CRUD | Not available | -| **Export** | `createMemoryExport` | Not available | -| **Feedback** | `feedback()` | Not available | -| **Project mgmt** | `getProject`, `updateProject` | Not available | -| **User listing** | `users()`, `deleteUser()` | Not available | -| **History** | Platform-managed | SQLite (configurable) | - ---- - -## v2 Compatibility - -If you're using SDK v2.x: - -**Naming Changes:** -- Top-level params now use camelCase: `topK`, `rerank` (not `top_k`) -- Filter keys use snake_case: `user_id`, `agent_id` -- OSS: `limit` renamed to `topK` - -**API Changes:** -```typescript -// v2 - top-level entity IDs, snake_case -await client.search("query", { user_id: "alice", top_k: 20 }); - -// v3 - filters object with snake_case keys, camelCase top-level params -await client.search("query", { filters: { user_id: "alice" }, topK: 20 }); -``` - -**Default Changes:** -| Param | v2 | v3 | -|-------|----|----| -| `topK` | 100 | 20 | -| `threshold` | none | 0.1 | -| `rerank` | true | false | - -**Removed:** -- `OutputFormat` and `API_VERSION` enums -- `organizationId`, `projectId` from constructor -- `enableGraph`, `asyncMode`, `outputFormat`, `immutable`, `expirationDate`, `filterMemories`, `batchSize`, `forceAddOnly`, `includes`, `excludes`, `keywordSearch` - -See the [v2 to v3 migration guide](https://docs.mem0.ai/migration/oss-v2-to-v3) for details. diff --git a/integrations/mem0-plugin/skills/mem0/client/python.md b/integrations/mem0-plugin/skills/mem0/client/python.md deleted file mode 100644 index 0cf35a550..000000000 --- a/integrations/mem0-plugin/skills/mem0/client/python.md +++ /dev/null @@ -1,487 +0,0 @@ -# Mem0 Python SDK Reference - -Complete reference for the `mem0ai` Python package. Covers both the Platform client (managed API) and the Open Source self-hosted variant. - ---- - -## Platform Client - -### Installation - -```bash -pip install mem0ai -export MEM0_API_KEY="m0-your-api-key" -``` - -### MemoryClient (Synchronous) - -```python -from mem0 import MemoryClient - -client = MemoryClient(api_key="m0-xxx") -``` - -**Constructor:** `MemoryClient(api_key=None)`. If `api_key` is not provided, reads from `MEM0_API_KEY` environment variable. Raises `ValueError` if no key found. - -- HTTP library: `httpx` -- Timeout: 300 seconds -- Base URL: `https://api.mem0.ai` - -### AsyncMemoryClient (Asynchronous) - -```python -from mem0 import AsyncMemoryClient - -client = AsyncMemoryClient(api_key="m0-xxx") - -# Or use as context manager -async with AsyncMemoryClient(api_key="m0-xxx") as client: - results = await client.search("query", filters={"user_id": "alice"}) -``` - -Same methods as `MemoryClient`, all `async`/`await`. Supports async context manager. - ---- - -### Memory Methods - -#### add(messages, **kwargs) - -Store new memories from messages. - -```python -messages = [ - {"role": "user", "content": "I'm a vegetarian and allergic to nuts."}, - {"role": "assistant", "content": "Got it! I'll remember that."} -] -client.add(messages, user_id="alice") -``` - -| Parameter | Type | Default | Description | -|-----------|------|---------|-------------| -| `messages` | str \| dict \| list[dict] | required | Message content. Strings auto-convert to user messages | -| `user_id` | str | None | User identifier | -| `agent_id` | str | None | Agent identifier | -| `app_id` | str | None | Application identifier | -| `run_id` | str | None | Session/run identifier | -| `metadata` | dict | None | Custom key-value pairs | -| `infer` | bool | True | If False, store raw text without LLM inference | -| `custom_categories` | list | None | Override project categories | -| `custom_instructions` | str | None | Override extraction instructions | -| `timestamp` | int \| float \| str | None | Custom timestamp (Unix epoch or ISO 8601) | - -**Returns:** `dict` -- list of events: `[{"id": "...", "event": "ADD", "data": {"memory": "..."}}]` - -#### search(query, **kwargs) - -Search memories by semantic similarity. - -```python -results = client.search("dietary preferences", filters={"user_id": "alice"}) -for mem in results.get("results", []): - print(mem["memory"], mem["score"]) -``` - -| Parameter | Type | Default | Description | -|-----------|------|---------|-------------| -| `query` | str | required | Natural language search query | -| `filters` | dict | None | Filter object with entity IDs and/or `AND`/`OR`/`NOT` conditions (e.g., `{"user_id": "alice"}`) | -| `top_k` | int | 10 | Number of results | -| `rerank` | bool | False | Enable deep semantic reranking (+150-200ms) | -| `threshold` | float | 0.1 | Minimum similarity score | -| `fields` | list | None | Specific fields to return | -| `categories` | list | None | Filter by category | - -**Returns:** `dict` -- `{"results": [{id, memory, user_id, categories, score, created_at, ...}]}` - -#### get(memory_id) - -Retrieve a single memory by ID. - -```python -memory = client.get(memory_id="ea925981-...") -``` - -**Returns:** `dict` -- full memory object - -#### get_all(**kwargs) - -Retrieve all memories with optional filtering. Requires at least one entity identifier. - -```python -memories = client.get_all(filters={"user_id": "alice"}) -# With compound filters -memories = client.get_all(filters={"AND": [{"user_id": "alice"}, {"categories": {"contains": "health"}}]}) -``` - -| Parameter | Type | Default | Description | -|-----------|------|---------|-------------| -| `filters` | dict | None | Filter object with entity IDs and/or `AND`/`OR`/`NOT` conditions | -| `top_k` | int | None | Limit results | -| `page` | int | None | Page number | -| `page_size` | int | None | Results per page | - -**Returns:** `dict` -- `{"results": [...]}` - -#### update(memory_id, text=None, metadata=None, timestamp=None) - -Update a memory's content, metadata, or timestamp. At least one parameter required. - -```python -client.update("ea925981-...", text="Updated: vegan since 2024") -client.update("ea925981-...", metadata={"verified": True}) -``` - -**Returns:** `dict` -- updated memory - -#### delete(memory_id) - -Permanently delete a single memory. - -```python -client.delete("ea925981-...") -``` - -#### delete_all(**kwargs) - -Delete all memories matching filters. Irreversible. - -```python -client.delete_all(user_id="alice") -``` - -#### history(memory_id) - -Get the change history of a memory. - -```python -history = client.history("ea925981-...") -# Returns: [{previous_value, new_value, action, timestamps}] -``` - ---- - -### Batch Methods - -#### batch_update(memories) - -Update up to 1000 memories in a single request. - -```python -client.batch_update([ - {"memory_id": "uuid-1", "text": "Updated text"}, - {"memory_id": "uuid-2", "text": "Another update", "metadata": {"verified": True}}, -]) -``` - -#### batch_delete(memories) - -Delete up to 1000 memories in a single request. - -```python -client.batch_delete([ - {"memory_id": "uuid-1"}, - {"memory_id": "uuid-2"}, -]) -``` - ---- - -### User/Entity Management - -#### users() - -List all users, agents, and sessions that have memories. - -```python -users = client.users() -# Returns: {"results": [{"type": "user", "name": "alice"}, ...]} -``` - -#### delete_users(user_id=None, agent_id=None, app_id=None, run_id=None) - -Delete a specific entity and all its memories. - -```python -client.delete_users(user_id="alice") -``` - -#### reset() - -Delete ALL users, agents, sessions, and memories. Complete data reset. - -```python -client.reset() -``` - ---- - -### Export & Summary - -#### create_memory_export(schema, **kwargs) - -Create a structured export of memories. - -```python -import json - -schema = json.dumps({ - "type": "object", - "properties": { - "name": {"type": "string"}, - "preferences": {"type": "array", "items": {"type": "string"}}, - } -}) -export = client.create_memory_export(schema=schema, user_id="alice") -``` - -#### get_memory_export(**kwargs) - -Retrieve a previously created export. - -```python -result = client.get_memory_export(memory_export_id=export["id"]) -``` - -#### get_summary(filters=None) - -Get a summary of memories. - -```python -summary = client.get_summary(filters={"user_id": "alice"}) -``` - ---- - -### Feedback - -#### feedback(memory_id, feedback=None, feedback_reason=None) - -Provide quality feedback on a memory. - -```python -client.feedback( - memory_id="mem-123", - feedback="POSITIVE", # POSITIVE | NEGATIVE | VERY_NEGATIVE | None (clear) - feedback_reason="Accurately captured preference" -) -``` - ---- - -### Webhooks - -```python -# List -webhooks = client.get_webhooks(project_id="proj_123") - -# Create -webhook = client.create_webhook( - url="https://your-app.com/webhook", - name="Memory Logger", - project_id="proj_123", - event_types=["memory_add", "memory_update"] -) - -# Update -client.update_webhook(webhook_id=123, name="Updated", url="https://new-url.com") - -# Delete -client.delete_webhook(webhook_id=123) -``` - ---- - -### Project Management - -Access via `client.project.*`: - -```python -# Get project config -config = client.project.get(fields=["custom_categories", "custom_instructions"]) - -# Update project settings -client.project.update( - custom_instructions="Extract dietary preferences and health info", - custom_categories=[{"health": "Medical and dietary info"}], - multilingual=True, -) - -# Create/delete project -client.project.create(name="My Project", description="...") -client.project.delete() - -# Member management -members = client.project.get_members() -client.project.add_member(email="user@example.com", role="READER") # READER or OWNER -client.project.update_member(email="user@example.com", role="OWNER") -client.project.remove_member(email="user@example.com") -``` - ---- - -## Open Source / Self-Hosted - -### Installation - -```bash -pip install mem0ai -``` - -### Memory Class - -```python -from mem0 import Memory - -m = Memory() # Uses default config (OpenAI embedder + in-memory vector store) -``` - -**Import:** `from mem0 import Memory` (NOT `MemoryClient` -- that is the Platform client) - -### Configuration - -```python -config = { - "llm": { - "provider": "openai", # openai, groq, azure, ollama, lmstudio, google, anthropic, mistral - "config": { - "model": "gpt-5-mini", - "api_key": "sk-xxx", - } - }, - "embedder": { - "provider": "openai", # openai, ollama, azure, lmstudio, google, huggingface - "config": { - "model": "text-embedding-3-small", - "api_key": "sk-xxx", - } - }, - "vector_store": { - "provider": "qdrant", # faiss, qdrant, pgvector, redis, supabase, azure_ai_search, memory - "config": { - "collection_name": "my_memories", - "host": "localhost", - "port": 6333, - } - }, - "history_db_path": "history.db", # SQLite path for change history - "custom_instructions": "...", # Custom LLM prompt for extraction -} - -m = Memory.from_config(config) -``` - -### Context Manager - -```python -with Memory(config) as m: - m.add("I prefer dark mode", user_id="alice") - results = m.search("preferences", filters={"user_id": "alice"}) -# SQLite connections released automatically -``` - -### Methods - -All methods mirror the Platform client but run locally: - -#### add(messages, *, user_id, agent_id, run_id, metadata, infer=True) - -```python -m.add("I'm a vegetarian", user_id="alice") -m.add([ - {"role": "user", "content": "I like hiking"}, - {"role": "assistant", "content": "Great outdoor activity!"} -], user_id="alice") -``` - -At least one of `user_id`, `agent_id`, `run_id` required. - -**Returns:** `{"results": [...], "relations": [...]}` - -#### search(query, *, filters=None, top_k=20, threshold=0.1, rerank=False) - -```python -results = m.search("dietary preferences", filters={"user_id": "alice"}, top_k=5) -``` - -Entity IDs (`user_id`, `agent_id`, `run_id`) must be passed inside the `filters` dict. - -Supports filter operators: `eq`, `ne`, `in`, `nin`, `gt`, `gte`, `lt`, `lte`, `contains`, `not_contains`. - -#### get(memory_id) / get_all(**kwargs) / update(memory_id, data, metadata=None) / delete(memory_id) / delete_all(**kwargs) / history(memory_id) - -Same interface as Platform client. - -#### reset() - -Clear the entire vector store collection and history database. Recreates the vector store. - -```python -m.reset() -``` - -#### close() - -Release SQLite connections. Called automatically when using context manager. - -### AsyncMemory - -```python -from mem0 import AsyncMemory - -m = AsyncMemory(config) -await m.add("text", user_id="alice") -results = await m.search("query", filters={"user_id": "alice"}) -``` - ---- - -## Key Differences: Platform vs OSS - -| Aspect | Platform (`MemoryClient`) | OSS (`Memory`) | -|--------|--------------------------|----------------| -| **Import** | `from mem0 import MemoryClient` | `from mem0 import Memory` | -| **Auth** | API key required (`MEM0_API_KEY`) | No API key -- config-based | -| **Execution** | API calls to `api.mem0.ai` | Local execution | -| **Infrastructure** | Fully managed | Self-managed vector DB, embedder, LLM | -| **Entity filtering** | `filters={"user_id": "..."}` | `filters={"user_id": "..."}` | -| **Batch ops** | `batch_update`, `batch_delete` | Not available | -| **Webhooks** | Full CRUD | Not available | -| **Export** | `create_memory_export`, `get_memory_export` | Not available | -| **Feedback** | `feedback()` | Not available | -| **Project mgmt** | `client.project.*` | Not available | -| **User listing** | `users()`, `delete_users()` | Not available | -| **Custom prompts** | Via project settings | Direct config (`custom_instructions`) | -| **History** | Platform-managed | SQLite (configurable) | -| **Async** | `AsyncMemoryClient` | `AsyncMemory` | - ---- - -## v2 Compatibility - -If you're using SDK v2.x or the v2 API: - -**API Changes:** -- **Entity IDs in search/get_all:** Pass `user_id`, `agent_id` as top-level kwargs instead of inside `filters` - ```python - # v2 - results = client.search("query", user_id="alice") - # v3 - results = client.search("query", filters={"user_id": "alice"}) - ``` -- **add() returns:** v2 returns ADD, UPDATE, DELETE events; v3 returns ADD only - -**Default Changes:** -| Param | v2 | v3 | -|-------|----|----| -| `top_k` | 100 | 20 | -| `threshold` | None | 0.1 | -| `rerank` | True | False | - -**Removed Parameters:** -- Constructor: `org_id`, `project_id` -- add(): `async_mode`, `output_format`, `enable_graph`, `immutable`, `expiration_date`, `filter_memories`, `batch_size`, `force_add_only`, `includes`, `excludes`, `keyword_search` -- search()/get_all(): `enable_graph` -- Config: `enable_graph`, `graph_store`, `custom_fact_extraction_prompt` (renamed to `custom_instructions`) - -See the [v2 to v3 migration guide](https://docs.mem0.ai/migration/oss-v2-to-v3) for full details. diff --git a/integrations/mem0-plugin/skills/mem0/references/api-reference.md b/integrations/mem0-plugin/skills/mem0/references/api-reference.md deleted file mode 100644 index 4ab8548ca..000000000 --- a/integrations/mem0-plugin/skills/mem0/references/api-reference.md +++ /dev/null @@ -1,150 +0,0 @@ -# Mem0 Platform API Reference - -REST API endpoints for the Mem0 Platform. Base URL: `https://api.mem0.ai` - -All endpoints require: `Authorization: Token ` - -## Endpoints - -| Operation | Method | URL | -|-----------|--------|-----| -| Add Memories | `POST` | `/v3/memories/add/` | -| Search Memories | `POST` | `/v3/memories/search/` | -| Get All Memories | `POST` | `/v3/memories/` | -| Get Single Memory | `GET` | `/v1/memories/{memory_id}/` | -| Update Memory | `PUT` | `/v1/memories/{memory_id}/` | -| Delete Memory | `DELETE` | `/v1/memories/{memory_id}/` | -| Delete All Memories | `DELETE` | `/v1/memories/?user_id=X&app_id=Y` | -| Get Event Status | `GET` | `/v1/event/{event_id}/` | - -## Memory Object Structure - -| Field | Type | Description | -|-------|------|-------------| -| `id` | string (UUID) | Unique memory identifier | -| `memory` | string | Text content of the memory | -| `user_id` | string | Associated user | -| `agent_id` | string (nullable) | Agent identifier | -| `app_id` | string (nullable) | Application identifier | -| `run_id` | string (nullable) | Run/session identifier | -| `metadata` | object | Custom key-value pairs | -| `categories` | array of strings | Auto-assigned category tags | -| `hash` | string | Content hash | -| `created_at` | datetime | Creation timestamp | -| `updated_at` | datetime | Last modification timestamp | - -Search results additionally include `score` (relevance metric). - -## Scoping Identifiers - -Memories can be scoped to different levels: - -| Scope | Parameter | Use Case | -|-------|-----------|----------| -| User | `user_id` | Per-user memory isolation | -| Agent | `agent_id` | Per-agent memory partitioning | -| Application | `app_id` | Cross-agent app-level memory | -| Run/Session | `run_id` | Session-scoped temporary memory | - -**Critical:** Combining `user_id` and `agent_id` in a single AND filter yields empty results. Entities are stored separately. Use `OR` logic or separate queries. - -## Processing Model - -- Memories are processed **asynchronously** (v3 default) -- Add responses return queued `ADD` events only (v3 is ADD-only, no UPDATE/DELETE) -- Poll status via `GET /v1/event/{event_id}/` - -## Filter System - -Filters use nested JSON with a logical operator at the root: - -```json -{ - "AND": [ - {"user_id": "alice"}, - {"categories": {"contains": "finance"}}, - {"created_at": {"gte": "2024-01-01"}} - ] -} -``` - -Root must be `AND`, `OR`, or `NOT`. Simple shorthand `{"user_id": "alice"}` also works. - -### Supported Operators - -| Operator | Description | -|----------|-------------| -| `eq` | Equal to (default) | -| `ne` | Not equal to | -| `in` | Matches any value in array | -| `gt`, `gte` | Greater than / greater than or equal | -| `lt`, `lte` | Less than / less than or equal | -| `contains` | Case-sensitive containment | -| `icontains` | Case-insensitive containment | -| `*` | Wildcard -- matches any non-null value | - -### Filterable Fields - -| Field | Valid Operators | -|-------|-----------------| -| `user_id`, `agent_id`, `app_id`, `run_id` | `eq`, `ne`, `in`, `*` | -| `created_at`, `updated_at`, `timestamp` | `gt`, `gte`, `lt`, `lte`, `eq`, `ne` | -| `categories` | `eq`, `ne`, `in`, `contains` | -| `metadata` | `eq`, `ne`, `contains` (top-level keys only) | -| `keywords` | `contains`, `icontains` | -| `memory_ids` | `in` | - -### Filter Constraints - -1. **Entity scope partitioning:** `user_id` AND `agent_id` in one `AND` block yields empty results. -2. **Metadata limitations:** Only top-level keys. Only `eq`, `contains`, `ne`. No `in` or `gt`. -3. **Operator syntax:** Use `gte`, `lt`, `ne`. SQL-style (`>=`, `!=`) rejected. -4. **Entity filter required for get-all:** At least one of `user_id`, `agent_id`, `app_id`, or `run_id`. -5. **Wildcard excludes null:** `*` matches only non-null values. -6. **Date format:** ISO 8601 (`YYYY-MM-DDTHH:MM:SSZ`). Timezone-naive defaults to UTC. - -## Response Formats - -### Add Response (v3) - -```json -{ - "message": "Memory processing has been queued for background execution", - "status": "PENDING", - "event_id": "evt-uuid" -} -``` - -v3 is ADD-only. No UPDATE or DELETE events. - -### Search Response - -```json -{ - "results": [ - { - "id": "ea925981-...", - "memory": "Is a vegetarian and allergic to nuts.", - "user_id": "user123", - "categories": ["food", "health"], - "score": 0.89, - "created_at": "2024-07-26T10:29:36.630547-07:00" - } - ] -} -``` - -In v3, `score` is a combined multi-signal relevance score. - -### Get All Response (v3) - -```json -{ - "count": 123, - "next": "https://api.mem0.ai/v3/memories/?page=2&page_size=50", - "previous": null, - "results": [...] -} -``` - -v3 returns paginated envelope. Use `page` and `page_size` query params. diff --git a/integrations/mem0-plugin/skills/mem0/references/architecture.md b/integrations/mem0-plugin/skills/mem0/references/architecture.md deleted file mode 100644 index 4a04c820b..000000000 --- a/integrations/mem0-plugin/skills/mem0/references/architecture.md +++ /dev/null @@ -1,330 +0,0 @@ -# Mem0 Platform Architecture - -How Mem0 processes, stores, and retrieves memories under the hood. - -## Table of Contents - -- [Core Concept](#core-concept) -- [Memory Processing Pipeline](#memory-processing-pipeline) -- [Retrieval Pipeline](#retrieval-pipeline) -- [Memory Lifecycle](#memory-lifecycle) -- [Memory Object Structure](#memory-object-structure) -- [Scoping & Multi-Tenancy](#scoping--multi-tenancy) -- [Memory Layers](#memory-layers) -- [Performance Characteristics](#performance-characteristics) - ---- - -## Core Concept - -Mem0 is a managed memory layer that sits between your AI application and users. Every integration follows the same 3-step loop: - -``` -User Input → Retrieve relevant memories → Enrich LLM prompt → Generate response → Store new memories -``` - -Mem0 handles the complexity of extraction, deduplication, conflict resolution, and semantic retrieval so your application only needs to call `search()` and `add()`. - -**Storage architecture:** -- **Vector store**: Embeddings for semantic similarity search -- **Entity store**: Automatic entity linking for relationship-aware retrieval - ---- - -## Memory Processing Pipeline - -### What happens when you call `client.add()` - -``` -Messages In - │ - ▼ -┌─────────────────────┐ -│ 1. EXTRACTION │ Single LLM call extracts all distinct new facts -│ (infer=True) │ If infer=False, stores raw text as-is -└─────────┬───────────┘ - │ - ▼ -┌─────────────────────┐ -│ 2. DEDUPLICATION │ Hash-based dedup (MD5 prevents exact duplicates) -│ │ No UPDATE/DELETE - v3 is ADD-only -└─────────┬───────────┘ - │ - ▼ -┌─────────────────────┐ -│ 3. STORAGE │ Batch embed → vector store -│ │ Entity extraction → entity store -└─────────┬───────────┘ - │ - ▼ - Memory Object -``` - -### Processing (v3) - -v3 processes memories asynchronously by default: -- API returns immediately: `{"status": "PENDING", "event_id": "evt-..."}` -- Poll status via `GET /v1/event/{event_id}/` -- Use webhooks for completion notifications - -### Extraction modes - -**Inferred (`infer=True`, default):** -- LLM extracts structured facts from conversation -- Conflict resolution deduplicates and resolves contradictions -- Best for: natural conversation → memory - -**Raw (`infer=False`):** -- Stores text exactly as provided, no LLM processing -- Skips conflict resolution — same fact can be stored twice -- Only `user` role messages are stored; `assistant` messages ignored -- Best for: bulk imports, pre-structured data, migrations - -**Warning:** Don't mix `infer=True` and `infer=False` for the same data — the same fact will be stored twice. - ---- - -## Retrieval Pipeline (v3) - -### What happens when you call `client.search()` - -``` -Query In - │ - ▼ -┌─────────────────────┐ -│ 1. PREPROCESSING │ Lemmatize keywords, extract entities -└─────────┬───────────┘ - │ - ▼ -┌─────────────────────┐ -│ 2. PARALLEL SCORING │ Semantic search (vector similarity) -│ │ BM25 keyword search (term matching) -│ │ Entity matching (entity graph boost) -└─────────┬───────────┘ - │ - ▼ -┌─────────────────────┐ -│ 3. SCORE FUSION │ Combine signals into single score -│ │ Optional: rerank=True for deep reordering -└─────────┬───────────┘ - │ - ▼ - Results (combined score per memory) -``` - -### v3 Search Defaults - -| Parameter | Default | Notes | -|-----------|---------|-------| -| `top_k` | 20 | Was 100 in v2 | -| `threshold` | 0.1 | Was None in v2 | -| `rerank` | False | Was True in v2 | - -### Implicit null scoping - -When you search with `filters={"user_id": "alice"}` only, Mem0 returns memories where `agent_id`, `app_id`, and `run_id` are all null. This prevents cross-scope leakage by default. - -To include memories with non-null fields, use explicit filters: -```python -# Gets memories for alice regardless of agent/app/run -filters={"OR": [{"user_id": "alice"}]} -``` - ---- - -## Memory Lifecycle (v3) - -v3 uses ADD-only extraction. Memories accumulate over time rather than being consolidated. - -### Creation -- `client.add(messages, user_id="...")` -- Single-pass extraction → deduplication → storage -- Returns `{"event_id": "...", "status": "PENDING"}` - -### Updates -- `client.update(memory_id, text="...")` replaces text -- Batch: `client.batch_update([...])` - -### Deletion -- Single: `client.delete(memory_id)` -- Batch: `client.batch_delete([...])` -- Bulk: `client.delete_all(filters={"user_id": "alice"})` - ---- - -## Memory Object Structure - -```json -{ - "id": "uuid-string", - "memory": "Extracted memory text", - "user_id": "user-identifier", - "agent_id": null, - "app_id": null, - "run_id": null, - "metadata": { "source": "chat", "priority": "high" }, - "categories": ["health", "preferences"], - "created_at": "2025-03-12T12:34:56Z", - "updated_at": "2025-03-12T12:34:56Z", - "structured_attributes": { - "day": 12, "month": 3, "year": 2025, - "hour": 12, "minute": 34, - "day_of_week": "wednesday", - "is_weekend": false, - "quarter": 1, "week_of_year": 11 - }, - "score": 0.85 -} -``` - -| Field | Type | Description | -|-------|------|-------------| -| `id` | UUID | Unique identifier, used for update/delete | -| `memory` | string | Extracted or stored text content | -| `user_id` | string | Primary entity scope | -| `agent_id` | string | Agent scope | -| `app_id` | string | Application scope | -| `run_id` | string | Session/run scope | -| `metadata` | object | Custom key-value pairs for filtering | -| `categories` | array | Auto-assigned or custom category tags | -| `created_at` | datetime | Creation timestamp | -| `updated_at` | datetime | Last modification timestamp | -| `structured_attributes` | object | Temporal breakdown for time-based queries | -| `score` | float | Semantic similarity (search results only, 0-1) | - ---- - -## Scoping & Multi-Tenancy - -Mem0 separates memories across four dimensions to prevent data mixing: - -| Dimension | Field | Purpose | Example | -|-----------|-------|---------|---------| -| User | `user_id` | Persistent persona or account | `"customer_6412"` | -| Agent | `agent_id` | Distinct agent or tool | `"meal_planner"` | -| App | `app_id` | Product surface or deployment | `"ios_retail_app"` | -| Session | `run_id` | Short-lived flow or thread | `"ticket-9241"` | - -### Storage model - -Each entity combination creates separate records. A memory with `user_id="alice"` is stored separately from one with `user_id="alice"` + `agent_id="bot"`. - -### Critical: cross-entity queries - -```python -# This returns NOTHING — user and agent memories are stored separately -filters={"AND": [{"user_id": "alice"}, {"agent_id": "bot"}]} - -# Use OR to query multiple scopes -filters={"OR": [{"user_id": "alice"}, {"agent_id": "bot"}]} - -# Use wildcard to include any non-null value -filters={"AND": [{"user_id": "*"}]} # All users (excludes null) -``` - -### Recommended scoping patterns - -```python -# User-level: persistent preferences -client.add(messages, user_id="alice") - -# Session-level: temporary context -client.add(messages, user_id="alice", run_id="session_123") -# Clean up when done: client.delete_all(run_id="session_123") - -# Agent-level: agent-specific knowledge -client.add(messages, agent_id="support_bot", app_id="helpdesk") - -# Multi-tenant: full isolation -client.add(messages, user_id="alice", agent_id="bot", app_id="acme_corp", run_id="ticket_42") -``` - ---- - -## Memory Layers - -Mem0 supports three layers of memory, from shortest to longest lived: - -### Conversation memory -- In-flight messages within a single turn -- Tool calls, chain-of-thought reasoning -- **Lifetime:** Single response — lost after turn finishes -- **Managed by:** Your application, not Mem0 - -### Session memory -- Short-lived facts for current task or channel -- Multi-step flows (onboarding, debugging, support tickets) -- **Lifetime:** Minutes to hours -- **Managed by:** Mem0 via `run_id` parameter -- Clean up with `client.delete_all(run_id="session_id")` - -### User memory -- Long-lived knowledge tied to a person or account -- Personal preferences, account state, compliance details -- **Lifetime:** Weeks to forever -- **Managed by:** Mem0 via `user_id` parameter -- Persists across all sessions and interactions - -### How layering works in practice - -```python -def chat(user_input: str, user_id: str, session_id: str) -> str: - # 1. Retrieve user memories (long-term preferences) - user_mems = mem0.search(user_input, filters={"user_id": user_id}) - - # 2. Retrieve session memories (current task context) - session_mems = mem0.search(user_input, filters={ - "AND": [{"user_id": user_id}, {"run_id": session_id}] - }) - - # 3. Combine both layers for LLM context - context = format_memories(user_mems) + format_memories(session_mems) - - # 4. Generate response - response = llm.generate(context=context, input=user_input) - - # 5. Store in session scope (temporary) + user scope (persistent) - messages = [{"role": "user", "content": user_input}, {"role": "assistant", "content": response}] - mem0.add(messages, user_id=user_id, run_id=session_id) - - return response -``` - ---- - -## Performance Characteristics - -### Latency - -| Operation | Typical Latency | -|-----------|----------------| -| Hybrid search (v3 default) | ~100-150ms | -| + reranking | +150-200ms | -| Add (async) | < 50ms response | - -### Processing - -- **Async (default):** Returns immediately, processes in background -- **Batch operations:** Up to 1000 memories per batch_update/batch_delete -- **Webhooks:** Real-time notifications when async processing completes - -### Scoping strategy for performance - -- Use `user_id` for all user-facing queries (most common, fastest) -- Add `run_id` for session isolation (narrows search space) -- Avoid wildcard `"*"` filters on large datasets (scans all non-null records) -- Use `top_k` to limit result count when you only need a few memories - ---- - -## Comparison with Alternatives - -| Approach | Pros | Cons | -|----------|------|------| -| **Raw vector DB** | Fast, full control | No extraction, no dedup, no conflict resolution | -| **In-memory chat history** | Zero latency | Lost on restart, no cross-session, grows unbounded | -| **RAG over documents** | Good for static knowledge | No personalization, no memory updates | -| **Mem0 Platform** | Managed extraction + dedup + graph + scoping | External dependency, async processing delay | - -Mem0 combines the best of vector search (semantic retrieval) with automatic extraction (LLM-powered), conflict resolution (deduplication), and structured scoping (multi-tenancy) — in a single managed API. diff --git a/integrations/mem0-plugin/skills/mem0/references/features.md b/integrations/mem0-plugin/skills/mem0/references/features.md deleted file mode 100644 index 1cfaba61f..000000000 --- a/integrations/mem0-plugin/skills/mem0/references/features.md +++ /dev/null @@ -1,406 +0,0 @@ -# Platform Features -- Mem0 Platform - -Additional platform capabilities beyond core CRUD operations. - -## Table of Contents - -- [Advanced Retrieval](#advanced-retrieval) -- [Entity Linking](#entity-linking) -- [Custom Categories](#custom-categories) -- [Custom Instructions](#custom-instructions) -- [Feedback Mechanism](#feedback-mechanism) -- [Memory Export](#memory-export) -- [Group Chat](#group-chat) -- [MCP Integration](#mcp-integration) -- [Webhooks](#webhooks) -- [Multimodal Support](#multimodal-support) - -## Advanced Retrieval - -### Hybrid Search (v3 Default) - -v3 uses multi-signal hybrid search combining: -- **Semantic search** (vector similarity) -- **BM25 keyword search** (normalized term matching) -- **Entity matching** (entity graph boost) - -This is automatic — no configuration needed. - -### Reranking (`rerank=True`) - -Deep semantic reordering of results — most relevant first. - -- Latency: +150-200ms -- Default: `False` (was `True` in v2) -- Best for: user-facing results, top-N precision - -**Python:** -```python -results = client.search(query, filters={"user_id": "user123"}, rerank=True) -``` - -**TypeScript:** -```typescript -const results = await client.search(query, { - filters: { user_id: 'user123' }, - rerank: true, -}); -``` - ---- - -## Entity Linking - -v3 replaces graph memory with built-in entity linking. Entities (proper nouns, quoted text, compound noun phrases) are automatically extracted and linked across memories. - -### How It Works - -1. **Extraction**: During `add()`, entities are automatically extracted from memory text -2. **Storage**: Entities are stored in a parallel collection (`{collection}_entities`) -3. **Retrieval**: During `search()`, query entities are matched and used to boost relevant memories - -Entity linking is automatic — no configuration required. The boost is folded into the combined `score` on each result. - -### v2 Migration Note - -If you were using `enable_graph=True` in v2: -- Remove `enable_graph` from all API calls -- Remove `graph_store` from OSS configuration -- Entity relationships are now consumed through retrieval ranking, not exposed as a separate `relations` array - -See the [v2 to v3 migration guide](https://docs.mem0.ai/migration/platform-v2-to-v3) for details. - ---- - -## Custom Categories - -Replace Mem0's default 15 labels with domain-specific categories. The system automatically tags memories to the closest matching category. - -### Default Categories (15) - -`personal_details`, `family`, `professional_details`, `sports`, `travel`, `food`, `music`, `health`, `technology`, `hobbies`, `fashion`, `entertainment`, `milestones`, `user_preferences`, `misc` - -### Configuration - -**Set project-level categories:** -```python -new_categories = [ - {"lifestyle_management": "Tracks daily routines, habits, wellness activities"}, - {"seeking_structure": "Documents goals around creating routines and systems"}, - {"personal_information": "Basic information about the user"} -] -client.project.update(custom_categories=new_categories) -``` - -```javascript -await client.updateProject({ customCategories: newCategories }); -``` - -**Retrieve active categories:** -```python -categories = client.project.get(fields=["custom_categories"]) -``` - -**Override categories for a single add call:** -```python -client.add(messages, user_id="alice", custom_categories=per_call_categories) -``` - -```javascript -await client.add(messages, { userId: "alice", customCategories: perCallCategories }); -``` - -### Resolution Order - -1. `custom_categories` passed on the `add` call -2. `custom_categories` set on the project -3. Built-in default catalog - -### Key Constraints - -- A per-call list **fully replaces** the project list for that call. The lists are not merged. -- Categories are applied at ingestion time. Changing the list later does not re-tag existing memories. - -### Main Use Case - -Per-call lists give different users or entities their own vocabulary inside a single project, without splitting them across projects. - ---- - -## Custom Instructions - -Natural language filters that control what information Mem0 extracts when creating memories. - -### Set Instructions - -```python -client.project.update(custom_instructions="Your guidelines here...") -``` - -```javascript -await client.updateProject({ customInstructions: "Your guidelines here..." }); -``` - -### Template Structure - -1. **Task Description** -- brief extraction overview -2. **Information Categories** -- numbered sections with specific details to capture -3. **Processing Guidelines** -- quality and handling rules -4. **Exclusion List** -- sensitive/irrelevant data to filter out - -### Domain Examples - -**E-commerce:** Capture product issues, preferences, service experience; exclude payment data. - -**Education:** Extract learning progress, student preferences, performance patterns; exclude specific grades. - -**Finance:** Track financial goals, life events, investment interests; exclude account numbers and SSNs. - -### Best Practices - -- Start simply, test with sample messages, iterate based on results -- Avoid overly lengthy instructions -- Be specific about what to include AND exclude - ---- - -## Feedback Mechanism - -Provide feedback on extracted memories to improve system quality over time. - -### Feedback Types - -| Type | Meaning | -|------|---------| -| `POSITIVE` | Memory is useful and accurate | -| `NEGATIVE` | Memory is not useful | -| `VERY_NEGATIVE` | Memory is harmful or completely wrong | -| `None` | Clear existing feedback | - -### Usage - -**Python:** -```python -client.feedback( - memory_id="mem-123", - feedback="POSITIVE", - feedback_reason="Accurately captured dietary preference" -) - -# Bulk feedback -for item in feedback_data: - client.feedback(**item) -``` - -**TypeScript:** -```typescript -await client.feedback('mem-123', { - feedback: 'POSITIVE', - feedbackReason: 'Accurately captured dietary preference', -}); -``` - ---- - -## Memory Export - -Create structured exports of memories using customizable schemas with filters. - -### Usage - -```python -import json - -# Define export schema -schema = { - "type": "object", - "properties": { - "name": {"type": "string"}, - "preferences": {"type": "array", "items": {"type": "string"}}, - "health_info": {"type": "string"}, - } -} - -# Create export -response = client.create_memory_export( - schema=json.dumps(schema), - filters={"user_id": "alice"}, - export_instructions="Create comprehensive profile based on all memories" -) - -# Retrieve export (may take a moment to process) -result = client.get_memory_export(memory_export_id=response["id"]) -``` - -**Best for:** Data analytics, user profile generation, compliance audits, CRM sync. - ---- - -## Group Chat - -Process multi-participant conversations and automatically attribute memories to individual speakers. - -### Usage - -```python -messages = [ - {"role": "user", "name": "Alice", "content": "I think we should use React for the frontend"}, - {"role": "user", "name": "Bob", "content": "I prefer Vue.js, it's simpler for our use case"}, - {"role": "assistant", "content": "Both are great choices. Let me note your preferences."}, -] - -# Mem0 automatically attributes memories to each speaker -response = client.add(messages, run_id="team_meeting_1") - -# Retrieve Alice's memories from that session -alice_mems = client.get_all( - filters={"AND": [{"user_id": "alice"}, {"run_id": "team_meeting_1"}]} -) -``` - -Use the `name` field in messages to identify speakers. Mem0 maps names to entity scopes automatically. - ---- - -## MCP Integration - -Model Context Protocol integration enables AI clients (Claude, Claude Code, Cursor, Windsurf, VS Code, OpenCode) to manage Mem0 memory autonomously. - -### Setup - -Add Mem0 MCP to your clients with a single command: - -```bash -npx mcp-add \ - --name mem0-mcp \ - --type http \ - --url "https://mcp.mem0.ai/mcp" \ - --clients "claude,claude code,cursor,windsurf,vscode,opencode" -``` - -### Available MCP Tools - -The MCP server exposes 9 memory tools that AI agents can use autonomously: -- Add, search, get, update, delete memories -- Get history, list users, delete users -- Search Mem0 documentation - -### How It Works - -1. Add Mem0 MCP to your AI client using the setup command above -2. The agent autonomously decides when to store/retrieve memories -3. No manual API calls needed — the agent manages memory as part of its reasoning - -**Best for:** Universal AI client integration — one protocol works everywhere. - ---- - -## Webhooks - -Real-time event notifications for memory operations. - -### Supported Events - -| Event | Trigger | -|-------|---------| -| `memory_add` | Memory created | -| `memory_update` | Memory modified | -| `memory_delete` | Memory removed | -| `memory_categorize` | Memory tagged | - -### Create Webhook - -Note: `project_id` here refers to the Mem0 dashboard project scope for webhooks — not the deprecated client init parameter. - -```python -webhook = client.create_webhook( - url="https://your-app.com/webhook", - name="Memory Logger", - project_id="proj_123", - event_types=["memory_add", "memory_categorize"] -) -``` - -### Manage Webhooks - -```python -# Retrieve -webhooks = client.get_webhooks(project_id="proj_123") - -# Update -client.update_webhook( - name="Updated Logger", - url="https://your-app.com/new-webhook", - event_types=["memory_update", "memory_add"], - webhook_id="wh_123" -) - -# Delete -client.delete_webhook(webhook_id="wh_123") -``` - -### Payload Structure - -Memory events contain: ID, data object with memory content, event type (`ADD`/`UPDATE`/`DELETE`). -Categorization events contain: memory ID, event type (`CATEGORIZE`), assigned category labels. - ---- - -## Multimodal Support - -Mem0 can process images and documents alongside text. - -### Supported Media Types - -- Images: JPG, PNG -- Documents: MDX, TXT, PDF - -### Image via URL - -```python -image_message = { - "role": "user", - "content": { - "type": "image_url", - "image_url": {"url": "https://example.com/image.jpg"} - } -} -client.add([image_message], user_id="alice") -``` - -### Image via Base64 - -```python -import base64 -with open("photo.jpg", "rb") as f: - base64_image = base64.b64encode(f.read()).decode("utf-8") - -image_message = { - "role": "user", - "content": { - "type": "image_url", - "image_url": {"url": f"data:image/jpeg;base64,{base64_image}"} - } -} -client.add([image_message], user_id="alice") -``` - -### Document (MDX/TXT) - -```python -doc_message = { - "role": "user", - "content": {"type": "mdx_url", "mdx_url": {"url": document_url}} -} -client.add([doc_message], user_id="alice") -``` - -### PDF Document - -```python -pdf_message = { - "role": "user", - "content": {"type": "pdf_url", "pdf_url": {"url": pdf_url}} -} -client.add([pdf_message], user_id="alice") -``` diff --git a/integrations/mem0-plugin/skills/mem0/references/integration-patterns.md b/integrations/mem0-plugin/skills/mem0/references/integration-patterns.md deleted file mode 100644 index 71cfa981c..000000000 --- a/integrations/mem0-plugin/skills/mem0/references/integration-patterns.md +++ /dev/null @@ -1,395 +0,0 @@ -# Mem0 Integration Patterns - -Working code examples for integrating Mem0 Platform with popular AI frameworks. -All examples use `MemoryClient` (Platform API key). - -Code examples are sourced from official Mem0 integration docs at docs.mem0.ai, simplified for quick reference. - ---- - -## Common Pattern - -Every integration follows the same 3-step loop: - -1. **Retrieve** -- search relevant memories before generating a response -2. **Generate** -- include memories as context in the LLM prompt -3. **Store** -- save the interaction back to Mem0 for future use - ---- - -## LangChain - -Source: [docs.mem0.ai/integrations/langchain](https://docs.mem0.ai/integrations/langchain) - -```python -from langchain_openai import ChatOpenAI -from langchain_core.messages import SystemMessage, HumanMessage -from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder -from mem0 import MemoryClient - -llm = ChatOpenAI(model="gpt-5-mini") -mem0 = MemoryClient() - -prompt = ChatPromptTemplate.from_messages([ - SystemMessage(content="You are a helpful travel agent AI. Use the provided context to personalize your responses."), - MessagesPlaceholder(variable_name="context"), - HumanMessage(content="{input}") -]) - -def retrieve_context(query: str, user_id: str): - """Retrieve relevant memories from Mem0""" - memories = mem0.search(query, filters={"user_id": user_id}) - memory_list = memories['results'] - serialized = ' '.join([m["memory"] for m in memory_list]) - return [ - {"role": "system", "content": f"Relevant information: {serialized}"}, - {"role": "user", "content": query} - ] - -def chat_turn(user_input: str, user_id: str) -> str: - # 1. Retrieve - context = retrieve_context(user_input, user_id) - # 2. Generate - chain = prompt | llm - response = chain.invoke({"context": context, "input": user_input}) - # 3. Store - mem0.add( - [{"role": "user", "content": user_input}, {"role": "assistant", "content": response.content}], - user_id=user_id - ) - return response.content -``` - ---- - -## CrewAI - -Source: [docs.mem0.ai/integrations/crewai](https://docs.mem0.ai/integrations/crewai) - -CrewAI has native Mem0 integration via `memory_config`: - -```python -from crewai import Agent, Task, Crew, Process -from mem0 import MemoryClient - -client = MemoryClient() - -# Store user preferences first -messages = [ - {"role": "user", "content": "I am more of a beach person than a mountain person."}, - {"role": "assistant", "content": "Noted! I'll recommend beach destinations."}, - {"role": "user", "content": "I like Airbnb more than hotels."}, -] -client.add(messages, user_id="crew_user_1") - -# Create agent -travel_agent = Agent( - role="Personalized Travel Planner", - goal="Plan personalized travel itineraries", - backstory="You are a seasoned travel planner.", - memory=True, -) - -# Create task -task = Task( - description="Find places to live, eat, and visit in San Francisco.", - expected_output="A detailed list of places to live, eat, and visit.", - agent=travel_agent, -) - -# Setup crew with Mem0 memory -crew = Crew( - agents=[travel_agent], - tasks=[task], - process=Process.sequential, - memory=True, - memory_config={ - "provider": "mem0", - "config": {"user_id": "crew_user_1"}, - } -) - -result = crew.kickoff() -``` - ---- - -## Vercel AI SDK - -> **Dedicated skill available.** For comprehensive Vercel AI SDK documentation, see the [mem0-vercel-ai-sdk skill](https://github.com/mem0ai/mem0/tree/main/skills/mem0-vercel-ai-sdk). - -Install: `npm install @mem0/vercel-ai-provider` - -Quick example (wrapped model with automatic memory): - -```typescript -import { generateText } from "ai"; -import { createMem0 } from "@mem0/vercel-ai-provider"; - -const mem0 = createMem0(); -const { text } = await generateText({ - model: mem0("gpt-5-mini", { user_id: "borat" }), - prompt: "Suggest me a good car to buy!", -}); -``` - -Supported providers: `openai`, `anthropic`, `google`, `groq`, `cohere` - ---- - -## OpenAI Agents SDK - -Source: [docs.mem0.ai/integrations/openai-agents-sdk](https://docs.mem0.ai/integrations/openai-agents-sdk) - -```python -from agents import Agent, Runner, function_tool -from mem0 import MemoryClient - -mem0 = MemoryClient() - -@function_tool -def search_memory(query: str, user_id: str) -> str: - """Search through past conversations and memories""" - memories = mem0.search(query, filters={"user_id": user_id}, top_k=3) - if memories and memories.get('results'): - return "\n".join([f"- {mem['memory']}" for mem in memories['results']]) - return "No relevant memories found." - -@function_tool -def save_memory(content: str, user_id: str) -> str: - """Save important information to memory""" - mem0.add([{"role": "user", "content": content}], user_id=user_id) - return "Information saved to memory." - -agent = Agent( - name="Personal Assistant", - instructions="""You are a helpful personal assistant with memory capabilities. - Use search_memory to recall past conversations. - Use save_memory to store important information.""", - tools=[search_memory, save_memory], - model="gpt-5-mini" -) - -result = Runner.run_sync(agent, "I love Italian food and I'm planning a trip to Rome next month") -print(result.final_output) -``` - -### Multi-Agent with Handoffs - -```python -from agents import Agent, Runner, function_tool - -travel_agent = Agent( - name="Travel Planner", - instructions="You are a travel planning specialist. Use search_memory and save_memory tools.", - tools=[search_memory, save_memory], - model="gpt-5-mini" -) - -health_agent = Agent( - name="Health Advisor", - instructions="You are a health and wellness advisor. Use search_memory and save_memory tools.", - tools=[search_memory, save_memory], - model="gpt-5-mini" -) - -triage_agent = Agent( - name="Personal Assistant", - instructions="""Route travel questions to Travel Planner, health questions to Health Advisor.""", - handoffs=[travel_agent, health_agent], - model="gpt-5-mini" -) - -result = Runner.run_sync(triage_agent, "Plan a healthy meal for my Italy trip") -``` - ---- - -## Pipecat (Voice / Real-Time) - -Source: [docs.mem0.ai/integrations/pipecat](https://docs.mem0.ai/integrations/pipecat) - -```python -from pipecat.services.mem0 import Mem0MemoryService - -memory = Mem0MemoryService( - api_key=os.getenv("MEM0_API_KEY"), - user_id="alice", - agent_id="voice_bot", - params={ - "search_limit": 10, - "search_threshold": 0.1, - "system_prompt": "Here are your past memories:", - "add_as_system_message": True, - } -) - -# Use in pipeline -pipeline = Pipeline([ - transport.input(), - stt, - user_context, - memory, # Memory enhances context automatically - llm, - transport.output(), - assistant_context -]) -``` - - - ---- - -## LangGraph - -Source: [docs.mem0.ai/integrations/langgraph](https://docs.mem0.ai/integrations/langgraph) - -State-based agent workflows with memory persistence. Best for complex conversation flows with branching logic. - -```python -from typing import Annotated, TypedDict, List -from langgraph.graph import StateGraph, START -from langgraph.graph.message import add_messages -from langchain_openai import ChatOpenAI -from mem0 import MemoryClient -from langchain_core.messages import SystemMessage, HumanMessage, AIMessage - -llm = ChatOpenAI(model="gpt-5-mini") -mem0 = MemoryClient() - -class State(TypedDict): - messages: Annotated[List[HumanMessage | AIMessage], add_messages] - mem0_user_id: str - -def chatbot(state: State): - messages = state["messages"] - user_id = state["mem0_user_id"] - - # Retrieve relevant memories - memories = mem0.search(messages[-1].content, filters={"user_id": user_id}) - context = "Relevant context:\n" - for memory in memories["results"]: - context += f"- {memory['memory']}\n" - - system_message = SystemMessage(content=f"""You are a helpful support assistant. -{context}""") - - response = llm.invoke([system_message] + messages) - - # Store the interaction - mem0.add( - [{"role": "user", "content": messages[-1].content}, - {"role": "assistant", "content": response.content}], - user_id=user_id - ) - return {"messages": [response]} - -graph = StateGraph(State) -graph.add_node("chatbot", chatbot) -graph.add_edge(START, "chatbot") -app = graph.compile() - -# Usage -result = app.invoke({ - "messages": [HumanMessage(content="I need help with my order")], - "mem0_user_id": "customer_123" -}) -``` - ---- - -## LlamaIndex - -Source: [docs.mem0.ai/integrations/llama-index](https://docs.mem0.ai/integrations/llama-index) - -Install: `pip install llama-index-core llama-index-memory-mem0` - -LlamaIndex has native Mem0 support via `Mem0Memory`. Works with ReAct and FunctionCalling agents. - -```python -from llama_index.memory.mem0 import Mem0Memory - -context = {"user_id": "alice", "agent_id": "llama_agent_1"} -memory = Mem0Memory.from_client( - context=context, - search_msg_limit=4, # messages from chat history used for retrieval (default: 5) -) - -# Use with LlamaIndex agent -from llama_index.core.agent import FunctionCallingAgent -from llama_index.llms.openai import OpenAI - -llm = OpenAI(model="gpt-5-mini") -agent = FunctionCallingAgent.from_tools( - tools=[], - llm=llm, - memory=memory, - verbose=True, -) - -response = agent.chat("I prefer vegetarian restaurants") -# Memory automatically stores and retrieves context -response = agent.chat("What kind of food do I like?") -# Agent retrieves the vegetarian preference from Mem0 -``` - ---- - -## AutoGen - -Source: [docs.mem0.ai/integrations/autogen](https://docs.mem0.ai/integrations/autogen) - -Install: `pip install autogen mem0ai` - -Multi-agent conversational systems with memory persistence. - -```python -from autogen import ConversableAgent -from mem0 import MemoryClient - -memory_client = MemoryClient() -USER_ID = "alice" - -agent = ConversableAgent( - "chatbot", - llm_config={"config_list": [{"model": "gpt-5-mini", "api_key": os.environ["OPENAI_API_KEY"]}]}, - code_execution_config=False, - human_input_mode="NEVER", -) - -def get_context_aware_response(question: str) -> str: - # Retrieve memories for context - relevant_memories = memory_client.search(question, filters={"user_id": USER_ID}) - context = "\n".join([m["memory"] for m in relevant_memories.get("results", [])]) - - prompt = f"""Answer considering previous interactions: - Previous context: {context} - Question: {question}""" - - reply = agent.generate_reply(messages=[{"content": prompt, "role": "user"}]) - - # Store the new interaction - memory_client.add( - [{"role": "user", "content": question}, {"role": "assistant", "content": reply}], - user_id=USER_ID - ) - return reply -``` - ---- - -## All Supported Frameworks - -Beyond the examples above, Mem0 integrates with: - -| Framework | Type | Install | -|-----------|------|---------| -| [Mastra](https://docs.mem0.ai/integrations/mastra) | TS agent framework | `npm install @mastra/mem0` | -| [ElevenLabs](https://docs.mem0.ai/integrations/elevenlabs) | Voice AI | `pip install elevenlabs mem0ai` | -| [LiveKit](https://docs.mem0.ai/integrations/livekit) | Real-time voice/video | `pip install livekit-agents mem0ai` | -| [Camel AI](https://docs.mem0.ai/integrations/camel-ai) | Multi-agent framework | `pip install camel-ai[all] mem0ai` | -| [AWS Bedrock](https://docs.mem0.ai/integrations/aws-bedrock) | Cloud LLM provider | `pip install boto3 mem0ai` | -| [Dify](https://docs.mem0.ai/integrations/dify) | Low-code AI platform | Plugin-based | -| [Google AI ADK](https://docs.mem0.ai/integrations/google-ai-adk) | Google agent framework | `pip install google-adk mem0ai` | - -For the general Python pattern (no framework), see the "Common integration pattern" in [SKILL.md](../SKILL.md). diff --git a/integrations/mem0-plugin/skills/mem0/references/quickstart.md b/integrations/mem0-plugin/skills/mem0/references/quickstart.md deleted file mode 100644 index 132982a49..000000000 --- a/integrations/mem0-plugin/skills/mem0/references/quickstart.md +++ /dev/null @@ -1,119 +0,0 @@ -# Mem0 Platform Quickstart - -Get running with Mem0 in 2 minutes. No infrastructure to deploy -- just an API key. - -## Prerequisites - -- Python 3.10+ or Node.js 18+ -- A Mem0 Platform API key ([Get one here](https://app.mem0.ai/dashboard/api-keys?utm_source=oss&utm_medium=mem0-plugin-skill-quickstart)) - -## Python Setup - -```bash -pip install mem0ai -export MEM0_API_KEY="m0-your-api-key" -``` - -```python -from mem0 import MemoryClient - -client = MemoryClient(api_key="your-api-key") - -# Add a memory -messages = [ - {"role": "user", "content": "I'm a vegetarian and allergic to nuts."}, - {"role": "assistant", "content": "Got it! I'll remember your dietary preferences."} -] -client.add(messages, user_id="user123") - -# Search memories -results = client.search("What are my dietary restrictions?", filters={"user_id": "user123"}) -print(results) -``` - -### Async Client - -```python -from mem0 import AsyncMemoryClient - -client = AsyncMemoryClient(api_key="your-api-key") - -await client.add(messages, user_id="user123") -results = await client.search("query", filters={"user_id": "user123"}) -``` - -## TypeScript / JavaScript Setup - -```bash -npm install mem0ai -export MEM0_API_KEY="m0-your-api-key" -``` - -```javascript -import MemoryClient from 'mem0ai'; - -const client = new MemoryClient({ apiKey: 'your-api-key' }); - -// Add a memory -const messages = [ - {"role": "user", "content": "I'm a vegetarian and allergic to nuts."}, - {"role": "assistant", "content": "Got it! I'll remember your dietary preferences."} -]; -await client.add(messages, { userId: "user123" }); - -// Search memories -const results = await client.search("What are my dietary restrictions?", { - filters: { user_id: "user123" } -}); -console.log(results); -``` - -## cURL - -```bash -export MEM0_API_KEY="m0-your-api-key" - -# Add memory -curl -X POST https://api.mem0.ai/v3/memories/add/ \ - -H "Authorization: Token $MEM0_API_KEY" \ - -H "Content-Type: application/json" \ - -d '{ - "messages": [ - {"role": "user", "content": "I am a vegetarian and allergic to nuts."}, - {"role": "assistant", "content": "Got it! I will remember your dietary preferences."} - ], - "user_id": "user123" - }' - -# Search memories -curl -X POST https://api.mem0.ai/v3/memories/search/ \ - -H "Authorization: Token $MEM0_API_KEY" \ - -H "Content-Type: application/json" \ - -d '{ - "query": "What are my dietary restrictions?", - "filters": {"user_id": "user123"} - }' -``` - -## Sample Response - -```json -{ - "results": [ - { - "id": "14e1b28a-2014-40ad-ac42-69c9ef42193d", - "memory": "Allergic to nuts", - "user_id": "user123", - "categories": ["health"], - "created_at": "2025-10-22T04:40:22.864647-07:00", - "score": 0.30 - } - ] -} -``` - -## Next Steps - -- [SDK Guide](sdk-guide.md) -- all methods for Python and TypeScript -- [API Reference](api-reference.md) -- REST endpoints and memory object structure -- [Integration Patterns](integration-patterns.md) -- LangChain, CrewAI, Vercel AI, etc. diff --git a/integrations/mem0-plugin/skills/mem0/references/sdk-guide.md b/integrations/mem0-plugin/skills/mem0/references/sdk-guide.md deleted file mode 100644 index 512c0d19c..000000000 --- a/integrations/mem0-plugin/skills/mem0/references/sdk-guide.md +++ /dev/null @@ -1,353 +0,0 @@ -# Mem0 SDK Guide - -Complete SDK reference for Python and TypeScript. All methods use `MemoryClient` (Platform API). - -> **For language-specific deep references (including OSS):** See [client/python.md](../client/python.md) and [client/node.md](../client/node.md). For Python vs TypeScript differences: [client/differences.md](../client/differences.md). - -## Initialization - -**Python:** -```python -from mem0 import MemoryClient -client = MemoryClient(api_key="m0-your-api-key") -``` - -**Python (Async):** -```python -from mem0 import AsyncMemoryClient -client = AsyncMemoryClient(api_key="m0-your-api-key") -``` - -**TypeScript:** -```typescript -import MemoryClient from 'mem0ai'; -const client = new MemoryClient({ apiKey: 'm0-your-api-key' }); -``` - -Constructor accepts `apiKey` (required) and `host` (optional, default: `https://api.mem0.ai`). - ---- - -## add() -- Store Memories - -**Python:** -```python -messages = [ - {"role": "user", "content": "I'm a vegetarian and allergic to nuts."}, - {"role": "assistant", "content": "Got it! I'll remember that."} -] -client.add(messages, user_id="alice") - -# With metadata -client.add(messages, user_id="alice", metadata={"source": "onboarding"}) -``` - -**TypeScript:** -```typescript -await client.add(messages, { userId: "alice" }); -await client.add(messages, { userId: "alice", metadata: { source: "onboarding" } }); -``` - -### Parameters - -| Name | Type | Description | -|------|------|-------------| -| `messages` | array | `[{"role": "user", "content": "..."}]` | -| `user_id` | string | User identifier (recommended) | -| `agent_id` | string | Agent identifier | -| `run_id` | string | Session identifier | -| `metadata` | object | Custom key-value pairs | -| `infer` | boolean | If `false`, store raw text without inference (default: `true`) | - -### Advanced Add Options - -```python -# Agent + session scoping -client.add(messages, user_id="alice", agent_id="nutrition-agent", run_id="session-456") - -# Raw text -- skip LLM inference -client.add( - [{"role": "user", "content": "User prefers dark mode."}], - user_id="alice", - infer=False, -) -``` - ---- - -## search() -- Find Memories - -**Python:** -```python -results = client.search("dietary preferences?", filters={"user_id": "alice"}) - -# With filters and reranking -results = client.search( - query="work experience", - filters={"AND": [{"user_id": "alice"}, {"categories": {"contains": "professional_details"}}]}, - top_k=5, - rerank=True, - threshold=0.5 -) -``` - -**TypeScript:** -```typescript -const results = await client.search("dietary preferences", { filters: { user_id: "alice" } }); -const results = await client.search("work experience", { - filters: { AND: [{ user_id: "alice" }, { categories: { contains: "professional_details" } }] }, - topK: 5, - rerank: true, -}); -``` - -### Parameters - -| Name | Type | Description | -|------|------|-------------| -| `query` | string | Natural language search query | -| `filters` | object | Filter object (AND/OR operators). Use `{"user_id": "..."}` to filter by user | -| `top_k` | number | Number of results (default: 10 for Platform) | -| `rerank` | boolean | Enable reranking for better relevance (default: `false`) | -| `threshold` | number | Minimum similarity score (default: 0.1) | - -### Common Filter Patterns - -**Python:** -```python -# Single user filter -filters={"user_id": "alice"} - -# OR across agents -filters={"OR": [{"user_id": "alice"}, {"agent_id": {"in": ["travel-agent", "sports-agent"]}}]} - -# Category filtering (partial match) -filters={"AND": [{"user_id": "alice"}, {"categories": {"contains": "finance"}}]} - -# Category filtering (exact match) -filters={"AND": [{"user_id": "alice"}, {"categories": {"in": ["personal_information"]}}]} - -# Wildcard (match any non-null run) -filters={"AND": [{"user_id": "alice"}, {"run_id": "*"}]} - -# Date range -filters={"AND": [ - {"user_id": "alice"}, - {"created_at": {"gte": "2024-01-01T00:00:00Z"}}, - {"created_at": {"lt": "2024-02-01T00:00:00Z"}} -]} - -# Exclude categories with NOT -filters={"AND": [{"user_id": "user_123"}, {"NOT": {"categories": {"in": ["spam", "test"]}}}]} - -# Multi-dimensional query -filters={"AND": [ - {"user_id": "user_123"}, - {"keywords": {"icontains": "invoice"}}, - {"categories": {"in": ["finance"]}}, - {"created_at": {"gte": "2024-01-01T00:00:00Z"}} -]} -``` - -**TypeScript:** -```typescript -// Single user filter -filters: { user_id: "alice" } - -// OR across agents -filters: { OR: [{ user_id: "alice" }, { agent_id: { in: ["travel-agent", "sports-agent"] } }] } - -// Category filtering (partial match) -filters: { AND: [{ user_id: "alice" }, { categories: { contains: "finance" } }] } - -// Category filtering (exact match) -filters: { AND: [{ user_id: "alice" }, { categories: { in: ["personal_information"] } }] } -``` - ---- - -## get() / getAll() -- Retrieve Memories - -**Python:** -```python -# Single memory by ID -memory = client.get(memory_id="ea925981-...") - -# All memories for a user -memories = client.get_all(filters={"user_id": "alice"}) - -# With date range -memories = client.get_all( - filters={"AND": [ - {"user_id": "alex"}, - {"created_at": {"gte": "2024-07-01", "lte": "2024-07-31"}} - ]} -) -``` - -**TypeScript:** -```typescript -const memory = await client.get("ea925981-..."); -const memories = await client.getAll({ filters: { user_id: "alice" } }); -``` - -**Note:** `get_all` requires at least one of `user_id`, `agent_id`, `app_id`, or `run_id` in filters. - ---- - -## update() -- Modify Memories - -**Python:** -```python -client.update(memory_id="ea925981-...", text="Updated: vegan since 2024") -client.update(memory_id="ea925981-...", text="Updated", metadata={"verified": True}) -``` - -**TypeScript:** -```typescript -await client.update("ea925981-...", { text: "Updated: vegan since 2024" }); -``` - ---- - -## delete() / deleteAll() -- Remove Memories - -**Python:** -```python -client.delete(memory_id="ea925981-...") -client.delete_all(user_id="alice") # Irreversible bulk delete -``` - -**TypeScript:** -```typescript -await client.delete("ea925981-..."); -await client.deleteAll({ userId: "alice" }); -``` - ---- - -## history() -- Track Changes - -**Python:** -```python -history = client.history(memory_id="ea925981-...") -# Returns: [{previous_value, new_value, action, timestamps}] -``` - -**TypeScript:** -```typescript -const history = await client.history("ea925981-..."); -``` - ---- - -## Batch Operations (TypeScript) - -```typescript -// Batch update -await client.batchUpdate([ - { memoryId: "uuid-1", text: "Updated text" }, - { memoryId: "uuid-2", text: "Another updated text" }, -]); - -// Batch delete -await client.batchDelete(["uuid-1", "uuid-2", "uuid-3"]); -``` - ---- - -## Additional Methods - -```python -# List all users/agents/sessions with memories -users = client.users() - -# Delete a user/agent entity -client.delete_users(user_id="alice") - -# Submit feedback on a memory -client.feedback(memory_id="...", feedback="POSITIVE", feedback_reason="Accurate extraction") - -# Export memories -export = client.create_memory_export(filters={"AND": [{"user_id": "alice"}]}) -data = client.get_memory_export(memory_export_id=export["id"]) -``` - ---- - -## Common Pitfalls - -1. **Entity cross-filtering fails silently** -- `AND` with `user_id` + `agent_id` returns empty. Use `OR`. -2. **SQL operators rejected** -- use `gte`, `lt`, etc. Not `>=`, `<`. -3. **Metadata filtering is limited** -- only top-level keys with `eq`, `contains`, `ne`. -4. **Wildcard `*` excludes null** -- only matches non-null values. -5. **Default threshold is 0.1** -- increase for stricter matching. -6. **Async processing** -- memories process asynchronously. Wait 2-3s after `add()` before searching. - -## Naming Conventions - -Python uses `snake_case` everywhere (`user_id`, `memory_id`, `get_all`). TypeScript uses `camelCase` for methods (`getAll`, `deleteAll`, `batchUpdate`) and top-level parameters (`userId`, `topK`, `pageSize`), but filter keys use `snake_case` (`user_id`, `agent_id`). - ---- - -## v2 to v3 Migration - -### Breaking Changes in v3 - -**1. Entity IDs in search() and getAll()** - -v3 requires entity IDs (`user_id`, `agent_id`, `run_id`) inside `filters` instead of as top-level parameters: - -```python -# v2 (deprecated) -client.search("query", user_id="alice") -client.get_all(user_id="alice") - -# v3 -client.search("query", filters={"user_id": "alice"}) -client.get_all(filters={"user_id": "alice"}) -``` - -```typescript -// v2 (deprecated) -await client.search("query", { user_id: "alice" }); -await client.getAll({ user_id: "alice" }); - -// v3 -await client.search("query", { filters: { user_id: "alice" } }); -await client.getAll({ filters: { user_id: "alice" } }); -``` - -**2. TypeScript Parameter Naming** - -v3 TypeScript uses camelCase for all parameters: - -| v2 | v3 | -|----|-----| -| `user_id` | `userId` | -| `agent_id` | `agentId` | -| `run_id` | `runId` | -| `top_k` | `topK` | -| `page_size` | `pageSize` | - -**3. Default Values Changed** - -| Parameter | v2 Default | v3 Default | -|-----------|------------|------------| -| `threshold` | 0.3 | 0.1 | -| `rerank` | (not specified) | `false` | - -**4. Removed Parameters** - -The following parameters are no longer supported: - -| Parameter | Status | -|-----------|--------| -| `enable_graph` | Removed from add/search/getAll | -| `keyword_search` | Removed from search | -| `filter_memories` | Removed | -| `immutable` | Removed from add | -| `expiration_date` | Removed from add | -| `includes` | Removed from add | -| `excludes` | Removed from add | -| `async_mode` | Removed from add | diff --git a/integrations/mem0-plugin/skills/mem0/references/use-cases.md b/integrations/mem0-plugin/skills/mem0/references/use-cases.md deleted file mode 100644 index 5f3ba655d..000000000 --- a/integrations/mem0-plugin/skills/mem0/references/use-cases.md +++ /dev/null @@ -1,720 +0,0 @@ -# Mem0 Use Cases & Examples - -Real-world implementation patterns for Mem0 Platform. Each use case includes complete, runnable code in both Python and TypeScript. - -## Table of Contents - -- [Personalized AI Companion](#1-personalized-ai-companion) -- [Customer Support with Categories](#2-customer-support-with-categories) -- [Healthcare Coach](#3-healthcare-coach) -- [Content Creation Workflow](#4-content-creation-workflow) -- [Multi-Agent / Multi-Tenant](#5-multi-agent--multi-tenant) -- [Personalized Search](#6-personalized-search) -- [Email Intelligence](#7-email-intelligence) -- [Common Patterns Across Use Cases](#common-patterns-across-use-cases) - ---- - -## 1. Personalized AI Companion - -A fitness coach that remembers goals, preferences, and progress across sessions. Mem0 persists context across app restarts — no session state needed. - -### Implementation (Python) - -```python -from mem0 import MemoryClient -from openai import OpenAI - -mem0 = MemoryClient() -openai_client = OpenAI() - -def chat(user_input: str, user_id: str) -> str: - # 1. Retrieve relevant memories - memories = mem0.search(user_input, user_id=user_id) - context = "\n".join([f"- {m['memory']}" for m in memories.get("results", [])]) - - # 2. Generate response with memory context - system_prompt = f"""You are Ray, a personal fitness coach. -Use these known facts about the user to personalize your response: -{context if context else 'No prior context yet.'}""" - - response = openai_client.chat.completions.create( - model="gpt-5-mini", - messages=[ - {"role": "system", "content": system_prompt}, - {"role": "user", "content": user_input}, - ] - ) - reply = response.choices[0].message.content - - # 3. Store interaction for future context - mem0.add( - [{"role": "user", "content": user_input}, {"role": "assistant", "content": reply}], - user_id=user_id - ) - return reply - -# Usage -chat("I want to run a marathon in under 4 hours", user_id="max") -# Next day, app restarted: -chat("What should I focus on today?", user_id="max") -# Ray remembers the sub-4 marathon goal -``` - -### Implementation (TypeScript) - -```typescript -import MemoryClient from 'mem0ai'; -import OpenAI from 'openai'; - -const mem0 = new MemoryClient({ apiKey: process.env.MEM0_API_KEY! }); -const openai = new OpenAI(); - -async function chat(userInput: string, userId: string): Promise { - // 1. Retrieve relevant memories - const memories = await mem0.search(userInput, { filters: { user_id: userId } }); - const context = memories.results - ?.map((m: any) => `- ${m.memory}`) - .join('\n') || 'No prior context yet.'; - - // 2. Generate response with memory context - const response = await openai.chat.completions.create({ - model: 'gpt-5-mini', - messages: [ - { role: 'system', content: `You are Ray, a personal fitness coach.\nUser context:\n${context}` }, - { role: 'user', content: userInput }, - ], - }); - const reply = response.choices[0].message.content!; - - // 3. Store interaction - await mem0.add( - [{ role: 'user', content: userInput }, { role: 'assistant', content: reply }], - { userId: userId } - ); - return reply; -} -``` - -### Key Benefits - -- Context persists across app restarts — no session management needed -- Memories are automatically deduplicated and updated -- Works with any LLM provider (OpenAI, Anthropic, etc.) - -**Best for:** Fitness coaches, tutors, therapists — any assistant that needs to remember goals across sessions. - ---- - -## 2. Customer Support with Categories - -Auto-categorize support data so teams retrieve the right facts fast. Uses custom categories for structured retrieval. - -### Implementation (Python) - -```python -from mem0 import MemoryClient - -client = MemoryClient() - -# 1. Define categories at the project level (one-time setup) -custom_categories = [ - {"support_tickets": "Customer issues and resolutions"}, - {"account_info": "Account details and preferences"}, - {"billing": "Payment history and billing questions"}, - {"product_feedback": "Feature requests and feedback"}, -] -client.project.update(custom_categories=custom_categories) - -# 2. Store interactions — auto-classified into categories -def log_support_interaction(user_id: str, message: str, priority: str = "normal"): - client.add( - [{"role": "user", "content": message}], - user_id=user_id, - metadata={"priority": priority, "source": "support_chat"} - ) - -# 3. Retrieve by category -def get_billing_issues(user_id: str): - return client.get_all( - filters={ - "AND": [ - {"user_id": user_id}, - {"categories": {"in": ["billing"]}} - ] - } - ) - -def search_support_history(user_id: str, query: str): - return client.search( - query, - filters={ - "AND": [ - {"user_id": user_id}, - {"categories": {"contains": "support_tickets"}} - ] - }, - top_k=5 - ) - -# Usage -log_support_interaction("maria", "I was charged twice for last month's subscription", priority="high") -log_support_interaction("maria", "The dashboard is loading slowly on mobile") -billing = get_billing_issues("maria") # Returns only billing-related memories -``` - -### Implementation (TypeScript) - -```typescript -import MemoryClient from 'mem0ai'; - -const client = new MemoryClient({ apiKey: process.env.MEM0_API_KEY! }); - -// Setup categories (one-time) -await client.updateProject({ - custom_categories: [ - { support_tickets: 'Customer issues and resolutions' }, - { billing: 'Payment history and billing questions' }, - { product_feedback: 'Feature requests and feedback' }, - ], -}); - -async function logInteraction(userId: string, message: string, priority = 'normal') { - await client.add( - [{ role: 'user', content: message }], - { userId: userId, metadata: { priority, source: 'support_chat' } } - ); -} - -async function getBillingIssues(userId: string) { - return client.getAll({ - filters: { AND: [{ user_id: userId }, { categories: { in: ['billing'] } }] }, - }); -} -``` - -### Key Benefits - -- Automatic categorization — no manual tagging -- Filter by category for structured retrieval -- Metadata (`priority`, `source`) enables multi-dimensional queries - -**Best for:** Help desks, SaaS support, e-commerce — structured retrieval by category eliminates manual scanning. - ---- - -## 3. Healthcare Coach - -Guide patients with an assistant that remembers medical history. Uses high `threshold` for confident retrieval in safety-critical contexts. - -### Implementation (Python) - -```python -from mem0 import MemoryClient -from openai import OpenAI - -mem0 = MemoryClient() -openai_client = OpenAI() - -def save_patient_info(user_id: str, information: str): - mem0.add( - [{"role": "user", "content": information}], - user_id=user_id, - run_id="healthcare_session", - metadata={"type": "patient_information"} - ) - -def consult(user_id: str, question: str) -> str: - # High threshold for medical accuracy - memories = mem0.search(question, user_id=user_id, top_k=5, threshold=0.7) - context = "\n".join([f"- {m['memory']}" for m in memories.get("results", [])]) - - response = openai_client.chat.completions.create( - model="gpt-5-mini", - messages=[ - {"role": "system", "content": f"You are a health coach. Patient context:\n{context}"}, - {"role": "user", "content": question}, - ] - ) - reply = response.choices[0].message.content - - # Store the interaction - mem0.add( - [{"role": "user", "content": question}, {"role": "assistant", "content": reply}], - user_id=user_id, - run_id="healthcare_session", - ) - return reply - -# Usage -save_patient_info("alex", "I'm allergic to penicillin and take metformin for type 2 diabetes") -consult("alex", "Can I take amoxicillin for my sore throat?") -# Remembers penicillin allergy — amoxicillin is a penicillin-type antibiotic -``` - -### Implementation (TypeScript) - -```typescript -import MemoryClient from 'mem0ai'; -import OpenAI from 'openai'; - -const mem0 = new MemoryClient({ apiKey: process.env.MEM0_API_KEY! }); -const openai = new OpenAI(); - -async function savePatientInfo(userId: string, info: string) { - await mem0.add( - [{ role: 'user', content: info }], - { userId: userId, runId: 'healthcare_session', metadata: { type: 'patient_information' } } - ); -} - -async function consult(userId: string, question: string): Promise { - const memories = await mem0.search(question, { - filters: { user_id: userId }, - topK: 5, - threshold: 0.7, - }); - const context = memories.results?.map((m: any) => `- ${m.memory}`).join('\n') || ''; - - const response = await openai.chat.completions.create({ - model: 'gpt-5-mini', - messages: [ - { role: 'system', content: `You are a health coach. Patient context:\n${context}` }, - { role: 'user', content: question }, - ], - }); - const reply = response.choices[0].message.content!; - - await mem0.add( - [{ role: 'user', content: question }, { role: 'assistant', content: reply }], - { userId: userId, runId: 'healthcare_session' } - ); - return reply; -} -``` - -### Key Benefits - -- High threshold (0.7) ensures only confident matches for safety-critical retrieval -- Session scoping via `run_id` groups related health interactions -- Metadata tagging separates patient info from conversation history - -**Best for:** Telehealth, wellness apps, patient management — persistent health context across visits. - ---- - -## 4. Content Creation Workflow - -Store voice guidelines once and apply them across every draft. Uses `run_id` and `metadata` to scope writing preferences per session. - -### Implementation (Python) - -```python -from mem0 import MemoryClient -from openai import OpenAI - -mem0 = MemoryClient() -openai_client = OpenAI() - -def store_writing_preferences(user_id: str, preferences: str): - mem0.add( - [{"role": "user", "content": preferences}], - user_id=user_id, - run_id="editing_session", - metadata={"type": "preferences", "category": "writing_style"} - ) - -def draft_content(user_id: str, topic: str) -> str: - # Retrieve writing preferences - prefs = mem0.search( - "writing style preferences", - filters={"AND": [{"user_id": user_id}, {"run_id": "editing_session"}]} - ) - style_context = "\n".join([f"- {m['memory']}" for m in prefs.get("results", [])]) - - response = openai_client.chat.completions.create( - model="gpt-5-mini", - messages=[ - {"role": "system", "content": f"Write content matching these style preferences:\n{style_context}"}, - {"role": "user", "content": f"Write a blog post about: {topic}"}, - ] - ) - return response.choices[0].message.content - -# Usage -store_writing_preferences("writer_01", "I prefer short sentences. Active voice. No jargon. Use analogies.") -draft_content("writer_01", "Why AI memory matters for chatbots") -# Drafts content matching the stored voice guidelines -``` - -### Implementation (TypeScript) - -```typescript -import MemoryClient from 'mem0ai'; -import OpenAI from 'openai'; - -const mem0 = new MemoryClient({ apiKey: process.env.MEM0_API_KEY! }); -const openai = new OpenAI(); - -async function storePreferences(userId: string, preferences: string) { - await mem0.add( - [{ role: 'user', content: preferences }], - { userId: userId, runId: 'editing_session', metadata: { type: 'preferences' } } - ); -} - -async function draftContent(userId: string, topic: string): Promise { - const prefs = await mem0.search('writing style preferences', { - filters: { AND: [{ user_id: userId }, { run_id: 'editing_session' }] }, - }); - const styleContext = prefs.results?.map((m: any) => `- ${m.memory}`).join('\n') || ''; - - const response = await openai.chat.completions.create({ - model: 'gpt-5-mini', - messages: [ - { role: 'system', content: `Write content matching these preferences:\n${styleContext}` }, - { role: 'user', content: `Write a blog post about: ${topic}` }, - ], - }); - return response.choices[0].message.content!; -} -``` - -### Key Benefits - -- Voice consistency across all content without repeating guidelines -- Scoped sessions let you maintain different style profiles -- Preferences update automatically as you refine them - -**Best for:** Marketing teams, technical writers, agencies — consistent voice across all content. - ---- - -## 5. Multi-Agent / Multi-Tenant - -Keep memories separate using `user_id`, `agent_id`, `app_id`, and `run_id` scoping. Critical for multi-agent workflows and multi-tenant apps. - -### Implementation (Python) - -```python -from mem0 import MemoryClient - -client = MemoryClient() - -# Store memories scoped to user + agent + session -def store_scoped_memory(messages: list, user_id: str, agent_id: str, run_id: str, app_id: str): - client.add( - messages, - user_id=user_id, - agent_id=agent_id, - run_id=run_id, - app_id=app_id - ) - -# Query within a specific scope -def search_user_session(query: str, user_id: str, app_id: str, run_id: str): - """Search memories for a specific user within a specific session.""" - return client.search( - query, - filters={ - "AND": [ - {"user_id": user_id}, - {"app_id": app_id}, - {"run_id": run_id} - ] - } - ) - -def search_agent_knowledge(query: str, agent_id: str, app_id: str): - """Search all memories an agent has across all users.""" - return client.search( - query, - filters={ - "AND": [ - {"agent_id": agent_id}, - {"app_id": app_id} - ] - } - ) - -# Usage: Travel concierge app with multiple agents -store_scoped_memory( - [{"role": "user", "content": "I'm vegetarian and prefer window seats"}], - user_id="traveler_cam", - agent_id="travel_planner", - run_id="tokyo-2025", - app_id="concierge_app" -) - -# User-scoped query: "What does Cam prefer?" -user_mems = search_user_session("dietary restrictions?", "traveler_cam", "concierge_app", "tokyo-2025") - -# Agent-scoped query: "What do all travelers prefer?" (across users) -agent_mems = search_agent_knowledge("common dietary restrictions?", "travel_planner", "concierge_app") -``` - -### Implementation (TypeScript) - -```typescript -import MemoryClient from 'mem0ai'; - -const client = new MemoryClient({ apiKey: process.env.MEM0_API_KEY! }); - -async function storeScopedMemory( - messages: Array<{ role: string; content: string }>, - userId: string, agentId: string, runId: string, appId: string -) { - await client.add(messages, { - userId: userId, - agentId: agentId, - runId: runId, - appId: appId, - }); -} - -async function searchUserSession(query: string, userId: string, appId: string, runId: string) { - return client.search(query, { - filters: { AND: [{ user_id: userId }, { app_id: appId }, { run_id: runId }] }, - }); -} - -async function searchAgentKnowledge(query: string, agentId: string, appId: string) { - return client.search(query, { - filters: { AND: [{ agent_id: agentId }, { app_id: appId }] }, - }); -} -``` - -### Key Benefits - -- Full isolation between users, agents, sessions, and apps -- Query at any scope level — user, agent, session, or app-wide -- No memory leakage between tenants - -**Best for:** Multi-agent workflows, multi-tenant SaaS — proper isolation at every level. - ---- - -## 6. Personalized Search - -Blend real-time search results with personal context. Uses `custom_instructions` to infer preferences from queries. - -### Implementation (Python) - -```python -from mem0 import MemoryClient -from openai import OpenAI - -mem0 = MemoryClient() -openai_client = OpenAI() - -# One-time setup: configure Mem0 to infer from queries -mem0.project.update( - custom_instructions="""Infer user preferences and facts from their search queries. -Extract dietary preferences, location, interests, and purchase history.""" -) - -def personalized_search(user_id: str, query: str, search_results: list) -> str: - # Get user context from memory - memories = mem0.search(query, user_id=user_id, top_k=5) - user_context = "\n".join([f"- {m['memory']}" for m in memories.get("results", [])]) - - response = openai_client.chat.completions.create( - model="gpt-5-mini", - messages=[ - {"role": "system", "content": f"Personalize search results using user context:\n{user_context}"}, - {"role": "user", "content": f"Query: {query}\n\nSearch results:\n{search_results}"}, - ] - ) - reply = response.choices[0].message.content - - # Store the query to learn preferences over time - mem0.add( - [{"role": "user", "content": query}], - user_id=user_id - ) - return reply - -# Usage -personalized_search("user_42", "best restaurants nearby", ["Restaurant A", "Restaurant B"]) -# Over time, Mem0 learns: "user prefers vegetarian, lives in Austin" -# Future searches are automatically personalized -``` - -### Implementation (TypeScript) - -```typescript -import MemoryClient from 'mem0ai'; -import OpenAI from 'openai'; - -const mem0 = new MemoryClient({ apiKey: process.env.MEM0_API_KEY! }); -const openai = new OpenAI(); - -async function personalizedSearch(userId: string, query: string, searchResults: string[]): Promise { - const memories = await mem0.search(query, { filters: { user_id: userId }, topK: 5 }); - const context = memories.results?.map((m: any) => `- ${m.memory}`).join('\n') || ''; - - const response = await openai.chat.completions.create({ - model: 'gpt-5-mini', - messages: [ - { role: 'system', content: `Personalize results using user context:\n${context}` }, - { role: 'user', content: `Query: ${query}\nResults: ${searchResults.join(', ')}` }, - ], - }); - const reply = response.choices[0].message.content!; - - await mem0.add([{ role: 'user', content: query }], { userId: userId }); - return reply; -} -``` - -### Key Benefits - -- Learns preferences from queries automatically via `custom_instructions` -- Personalizes any search provider (Tavily, Google, Bing) -- Zero manual preference setup — improves over time - -**Best for:** Personalized search engines, recommendation systems — search results tailored to individual users. - ---- - -## 7. Email Intelligence - -Capture, categorize, and recall inbox threads using persistent memories with rich metadata. - -### Implementation (Python) - -```python -from mem0 import MemoryClient - -client = MemoryClient() - -def store_email(user_id: str, sender: str, subject: str, body: str, date: str): - client.add( - [{"role": "user", "content": f"Email from {sender}: {subject}\n\n{body}"}], - user_id=user_id, - metadata={"email_type": "incoming", "sender": sender, "subject": subject, "date": date} - ) - -def search_emails(user_id: str, query: str): - return client.search( - query, - filters={"AND": [{"user_id": user_id}, {"categories": {"contains": "email"}}]}, - top_k=10 - ) - -def get_emails_from_sender(user_id: str, sender: str): - return client.get_all( - filters={ - "AND": [ - {"user_id": user_id}, - {"metadata": {"contains": sender}} - ] - } - ) - -# Usage -store_email("alice", "bob@acme.com", "Q3 Budget Review", "Attached is the Q3 budget...", "2025-01-15") -store_email("alice", "carol@acme.com", "Sprint Planning", "Here are the priorities...", "2025-01-16") - -results = search_emails("alice", "budget discussions") -sender_emails = get_emails_from_sender("alice", "bob@acme.com") -``` - -### Implementation (TypeScript) - -```typescript -import MemoryClient from 'mem0ai'; - -const client = new MemoryClient({ apiKey: process.env.MEM0_API_KEY! }); - -async function storeEmail(userId: string, sender: string, subject: string, body: string, date: string) { - await client.add( - [{ role: 'user', content: `Email from ${sender}: ${subject}\n\n${body}` }], - { userId: userId, metadata: { email_type: 'incoming', sender, subject, date } } - ); -} - -async function searchEmails(userId: string, query: string) { - return client.search(query, { - filters: { AND: [{ user_id: userId }, { categories: { contains: 'email' } }] }, - topK: 10, - }); -} -``` - -### Key Benefits - -- Rich metadata enables multi-dimensional queries (sender, date, subject) -- Category filtering separates emails from other memory types -- Semantic search across all email content - -**Best for:** Inbox management, email automation — searchable email memories with metadata filtering. - ---- - -## Common Patterns Across Use Cases - -### Pattern 1: Retrieve → Generate → Store - -Every use case follows the same 3-step loop: - -```python -# 1. Retrieve relevant context -memories = mem0.search(user_input, user_id=user_id) -context = "\n".join([m["memory"] for m in memories.get("results", [])]) - -# 2. Generate with context -response = llm.generate(system_prompt=f"Context:\n{context}", user_input=user_input) - -# 3. Store the interaction -mem0.add( - [{"role": "user", "content": user_input}, {"role": "assistant", "content": response}], - user_id=user_id -) -``` - -### Pattern 2: Scope with Entity Identifiers - -Use `user_id`, `agent_id`, `app_id`, and `run_id` to isolate memories: - -```python -# User-level: personal preferences -client.add(messages, user_id="alice") - -# Session-level: conversation within one session -client.add(messages, user_id="alice", run_id="session_123") - -# Agent-level: agent-specific knowledge -client.add(messages, agent_id="support_bot", app_id="helpdesk") -``` - -### Pattern 3: Rich Metadata for Filtering - -Attach structured metadata for multi-dimensional queries: - -```python -# Store with metadata -client.add(messages, user_id="alice", metadata={"priority": "high", "source": "phone_call"}) - -# Filter by category + metadata -client.search("billing issues", filters={ - "AND": [{"user_id": "alice"}, {"categories": {"contains": "billing"}}] -}) -``` - -### Pattern 4: Custom Instructions for Domain-Specific Extraction - -Control what Mem0 extracts from conversations: - -```python -client.project.update( - custom_instructions="Extract medical conditions, medications, and allergies. Exclude billing info." -) -``` - ---- - -## More Examples - -For 30+ cookbooks with complete working code: [docs.mem0.ai/cookbooks](https://docs.mem0.ai/cookbooks) diff --git a/integrations/mem0-plugin/skills/mem0/scripts/mem0_doc_search.py b/integrations/mem0-plugin/skills/mem0/scripts/mem0_doc_search.py deleted file mode 100755 index a04c89afb..000000000 --- a/integrations/mem0-plugin/skills/mem0/scripts/mem0_doc_search.py +++ /dev/null @@ -1,220 +0,0 @@ -#!/usr/bin/env python3 -""" -Mem0 Documentation Search Agent (Mintlify-based) -On-demand search tool for querying Mem0 documentation without storing content locally. - -This tool leverages Mintlify's documentation structure to perform just-in-time -retrieval of technical information from docs.mem0.ai. - -Usage: - python mem0_doc_search.py --query "how to add graph memory" - python mem0_doc_search.py --query "filter syntax for categories" - python mem0_doc_search.py --page "/platform/features/graph-memory" - python mem0_doc_search.py --index - python mem0_doc_search.py --query "webhook events" --section platform - -Purpose: - - Avoid bloating local context with full documentation - - Enable just-in-time retrieval of technical details - - Query specific documentation pages on demand - - Search across the full Mem0 documentation site -""" - -import argparse -import json -import sys -import urllib.error -import urllib.parse -import urllib.request - -DOCS_BASE = "https://docs.mem0.ai" -SEARCH_ENDPOINT = f"{DOCS_BASE}/api/search" -LLMS_INDEX = f"{DOCS_BASE}/llms.txt" - -# Known documentation sections for targeted retrieval -SECTION_MAP = { - "platform": [ - "/platform/overview", - "/platform/quickstart", - "/platform/features", - "/platform/features/graph-memory", - "/platform/features/selective-memory", - "/platform/features/custom-categories", - "/platform/features/v2-memory-filters", - "/platform/features/async-client", - "/platform/features/webhooks", - "/platform/features/multimodal-support", - ], - "api": [ - "/api-reference/memory/add-memories", - "/api-reference/memory/v2-search-memories", - "/api-reference/memory/v2-get-memories", - "/api-reference/memory/get-memory", - "/api-reference/memory/update-memory", - "/api-reference/memory/delete-memory", - ], - "open-source": [ - "/open-source/overview", - "/open-source/python-quickstart", - "/open-source/node-quickstart", - "/open-source/features", - "/open-source/features/graph-memory", - "/open-source/features/rest-api", - "/open-source/configure-components", - ], - "sdks": [ - "/sdks/python", - "/sdks/js", - ], - "integrations": [ - "/integrations", - ], -} - - -def fetch_url(url: str) -> str: - """Fetch content from a URL.""" - req = urllib.request.Request(url, headers={"User-Agent": "Mem0DocSearchAgent/1.0"}) - try: - with urllib.request.urlopen(req, timeout=15) as resp: - return resp.read().decode("utf-8") - except urllib.error.HTTPError as e: - return f"HTTP Error {e.code}: {e.reason}" - except urllib.error.URLError as e: - return f"URL Error: {e.reason}" - - -def search_docs(query: str, section: str | None = None) -> dict: - """ - Search Mem0 documentation using Mintlify's search API. - Falls back to the llms.txt index for keyword matching if the API is unavailable. - """ - # Try Mintlify search API first - params = urllib.parse.urlencode({"query": query}) - search_url = f"{SEARCH_ENDPOINT}?{params}" - - try: - result = fetch_url(search_url) - data = json.loads(result) - if isinstance(data, dict) and data.get("results"): - results = data["results"] - if section and section in SECTION_MAP: - section_paths = SECTION_MAP[section] - results = [r for r in results if any(r.get("url", "").startswith(p) for p in section_paths)] - return {"source": "mintlify_search", "results": results} - except (json.JSONDecodeError, Exception): - pass - - # Fallback: search llms.txt index for matching URLs - index_content = fetch_url(LLMS_INDEX) - query_lower = query.lower() - matching_urls = [] - - for line in index_content.splitlines(): - line = line.strip() - if not line or line.startswith("#"): - continue - if query_lower in line.lower(): - matching_urls.append(line) - - if section and section in SECTION_MAP: - section_paths = SECTION_MAP[section] - matching_urls = [u for u in matching_urls if any(p in u for p in section_paths)] - - return { - "source": "llms_txt_index", - "query": query, - "matching_urls": matching_urls[:20], - "suggestion": "Fetch specific URLs for detailed content", - } - - -def fetch_page(page_path: str) -> dict: - """Fetch a specific documentation page.""" - url = f"{DOCS_BASE}{page_path}" if page_path.startswith("/") else page_path - content = fetch_url(url) - return {"url": url, "content": content[:10000], "truncated": len(content) > 10000} - - -def get_index() -> dict: - """Fetch the full documentation index from llms.txt.""" - content = fetch_url(LLMS_INDEX) - urls = [line.strip() for line in content.splitlines() if line.strip() and not line.startswith("#")] - return {"total_pages": len(urls), "urls": urls, "sections": list(SECTION_MAP.keys())} - - -def list_section(section: str) -> dict: - """List all known pages in a documentation section.""" - if section not in SECTION_MAP: - return {"error": f"Unknown section: {section}", "available": list(SECTION_MAP.keys())} - return { - "section": section, - "pages": [f"{DOCS_BASE}{p}" for p in SECTION_MAP[section]], - } - - -def main(): - parser = argparse.ArgumentParser(description="Search Mem0 documentation on demand") - parser.add_argument("--query", help="Search query for documentation") - parser.add_argument("--page", help="Fetch a specific page path (e.g., /platform/features/graph-memory)") - parser.add_argument("--index", action="store_true", help="Show full documentation index") - parser.add_argument("--section", help="Filter by section or list section pages") - parser.add_argument("--json", action="store_true", help="Output as JSON") - - args = parser.parse_args() - - if args.index: - result = get_index() - elif args.section and not args.query: - result = list_section(args.section) - elif args.page: - result = fetch_page(args.page) - elif args.query: - result = search_docs(args.query, section=args.section) - else: - parser.print_help() - sys.exit(1) - - if args.json: - print(json.dumps(result, indent=2)) - else: - if isinstance(result, dict): - if "results" in result: - print(f"Source: {result.get('source', 'unknown')}") - for r in result["results"]: - print(f" - {r.get('title', 'N/A')}: {r.get('url', 'N/A')}") - if r.get("description"): - print(f" {r['description'][:200]}") - elif "matching_urls" in result: - print(f"Source: {result['source']}") - print(f"Query: {result['query']}") - for url in result["matching_urls"]: - print(f" - {url}") - if result.get("suggestion"): - print(f"\n{result['suggestion']}") - elif "urls" in result: - print(f"Total documentation pages: {result['total_pages']}") - print(f"Sections: {', '.join(result['sections'])}") - for url in result["urls"][:30]: - print(f" - {url}") - if result["total_pages"] > 30: - print(f" ... and {result['total_pages'] - 30} more") - elif "pages" in result: - print(f"Section: {result['section']}") - for page in result["pages"]: - print(f" - {page}") - elif "content" in result: - print(f"URL: {result['url']}") - if result.get("truncated"): - print("[Content truncated to 10000 chars]") - print(result["content"]) - elif "error" in result: - print(f"Error: {result['error']}") - if result.get("available"): - print(f"Available sections: {', '.join(result['available'])}") - else: - print(json.dumps(result, indent=2)) - - -if __name__ == "__main__": - main() diff --git a/integrations/mem0-plugin/skills/memory-reviewer/SKILL.md b/integrations/mem0-plugin/skills/memory-reviewer/SKILL.md deleted file mode 100644 index 4cc7a5970..000000000 --- a/integrations/mem0-plugin/skills/memory-reviewer/SKILL.md +++ /dev/null @@ -1,59 +0,0 @@ ---- -name: memory-reviewer -description: Reviews stored memory quality by detecting duplicates, contradictions, and stale entries with actionable recommendations. Use when search results seem conflicting, before running dream consolidation, or for periodic memory hygiene audits. ---- - -# Memory Reviewer - -Audits memory quality for the active project. Finds duplicates, contradictions, and low-confidence entries. - -## When to use - -- User asks "check my memories", "memory quality", "any duplicates?" -- User runs `/mem0:memory-reviewer` directly -- After a session with 5+ memory writes (suggest proactively) -- After `/mem0:health --deep` identifies issues - -## Steps - -1. **Fetch all memories** for active project via `get_memories` with `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `page_size=200`. Paginate if needed — cap at 200 memories. - -2. **Group by `metadata.type`**. Common types: `decision`, `convention`, `anti_pattern`, `task_learning`, `project_profile`, `user_preference`, `session_state`. - -3. **Scan each group for issues:** - - | Issue | Detection method | - |---|---| - | **Near-duplicates** | >60% noun overlap within same type. Compare memory text after stripping stop words. | - | **Contradictions** | Opposing facts about same topic (e.g., "use PostgreSQL" vs "use MySQL" for same component) | - | **Low-confidence** | `metadata.confidence < 0.3` | - | **Missing type** | No `metadata.type` set | - | **Stale** | `created_at` older than 180 days with no updates | - -4. **Output compact summary:** - -``` -memory-reviewer: project= total= - duplicates: found - contradictions: found - low_confidence: found - untagged: found - stale: found -``` - -5. **If issues found**, list them with memory IDs: - -``` -Issues: - [duplicate] "" ≈ "" [mem0:, mem0:] - [contradiction] "" vs "" [mem0:, mem0:] - [low_conf] "" (confidence: 0.1) [mem0:] -``` - -6. **Suggest action**: "Run `/mem0:dream` to consolidate duplicates and resolve contradictions." - -## Constraints - -- **Read-only** — never modify or delete memories (that's `/mem0:dream`'s job) -- **Max 200 memories** per scan -- Report findings, let user decide on action diff --git a/integrations/mem0-plugin/skills/onboard/SKILL.md b/integrations/mem0-plugin/skills/onboard/SKILL.md deleted file mode 100644 index 896e917ec..000000000 --- a/integrations/mem0-plugin/skills/onboard/SKILL.md +++ /dev/null @@ -1,184 +0,0 @@ ---- -name: onboard -description: Sets up mem0 for a new project including API key configuration, MCP authentication, project file import, and coding categories. Use on first run in a new project, when API key needs updating, or to re-run initial setup after configuration changes. ---- - -# Mem0 Onboarding Wizard - -Run this wizard to set up the mem0 plugin for the current project. Complete in ~60 seconds. - -**IMPORTANT: Execute steps strictly in order (0 → 1 → 2 → 3 → 4 → 5 → 6). Each step depends on the previous one. Do NOT run steps in parallel or skip ahead. Complete one step fully before starting the next.** - -## Step 0: Ensure mem0ai SDK is installed - -The plugin installs the `mem0ai` Python SDK automatically on session start via a venv in `${CLAUDE_PLUGIN_DATA}/venv`. If Step 5 (categories) fails with an import error, run: - -```bash -"${CLAUDE_PLUGIN_ROOT}/scripts/ensure_deps.sh" -``` - -This is silent and idempotent — safe to run anytime. - -## Step 1: Set up API key - -Check if the API key is available from any source: - -```bash -[ -n "${MEM0_API_KEY:-${CLAUDE_PLUGIN_OPTION_API_KEY:-}}" ] && echo "SET" || echo "NOT_SET" -``` - -IMPORTANT: Never run `echo $MEM0_API_KEY` — that prints the secret in plaintext to the conversation log. - -### If API key IS set (output is "SET") - -Print: `- API key found.` and proceed to Step 2. - -### If API key is NOT set (output is "NOT_SET") - -Guide the user through API key setup. Show this message: - -``` -Step 1: Setting up API key. - -- API key not found. Let's set it up. - - 1. Get your API key from https://app.mem0.ai/dashboard/api-keys - - 2. Choose ONE method: - - Option A — CLI (shell profile): - echo 'export MEM0_API_KEY="m0-your-key-here"' >> ~/.zshrc - source ~/.zshrc - - Option B — Desktop app (local environment editor): - Click the environment dropdown next to the prompt box, - hover over "Local", click the gear icon, and add: - MEM0_API_KEY = m0-your-key-here - (Stored encrypted on your machine, applies to all local sessions) - - Note: The Desktop app does NOT inherit custom env vars from - shell profiles — it only reads PATH. Use Option B for Desktop. - - 3. Verify: - [ -n "${MEM0_API_KEY:-${CLAUDE_PLUGIN_OPTION_API_KEY:-}}" ] && echo "SET" || echo "NOT_SET" -``` - -After the user confirms, re-run the verify command. If NOT_SET, repeat. If SET, proceed to Step 2. - -## Step 2: MCP server connection - -First, check if MCP tools are already available using ToolSearch with query `"mem0 search_memories"`. The exact tool name varies by install method (may be `mcp__mem0__search_memories` or `mcp__plugin_mem0_mem0__search_memories`). - -**If MCP tools ARE found:** Print `- MCP already connected.` and proceed to Step 3. - -**If MCP tools are NOT found:** - -The MCP server authenticates using the `MEM0_API_KEY` set in Step 1. No OAuth or browser login is needed. - -1. Verify the API key is set (re-run the Step 1 check) -2. Check the plugin is installed: run `/plugins` and confirm `mem0` appears -3. Check the MCP server is listed: run `/mcp` and look for `mcp.mem0.ai` -4. If the server shows an error, ask the user to restart Claude Code and run `/mem0:onboard` again -5. If all checks pass but tools are still missing: "Restart Claude Code and run `/mem0:onboard` again." - -**STOP here** — do not proceed without MCP tools. - -## Step 3: Verify connectivity and show identity - -Call `search_memories` with `query="project setup"`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=1` to verify connectivity. - -Print: -``` -- Connected - user: - project: - branch: -``` - -If the search fails, troubleshoot the API key and MCP connection. - -## Step 4: Import project files - -Project files (CLAUDE.md, AGENTS.md, etc.) are automatically imported into mem0 when a session starts. This step verifies import status and triggers a re-import if needed. - -### 4a: Detect project files - -```bash -for f in CLAUDE.md AGENTS.md .cursorrules .windsurfrules mem0.md; do - [ -f "$f" ] && echo "FOUND: $f ($(wc -c < "$f") bytes)" -done || true -``` - -If no files found, print `- No project files found. Skipping import.` and proceed to Step 5. - -### 4b: Check and import - -Run auto_import in foreground to check status and import if needed: - -```bash -MEM0_DEBUG=1 MEM0_CWD="$PWD" python3 "${CLAUDE_PLUGIN_ROOT}/scripts/auto_import.py" -``` - -### 4c: Report to user - -Parse the auto_import output and print a user-friendly summary: - -- If output contains `Imported` lines: - ``` - - Importing project files into mem0... done. - file(s) imported ( chunks). These are stored verbatim for future context. - ``` -- If output contains only `skipping` lines: - ``` - - Project files already in mem0 (imported during session start). Verified server-side. - ``` -- If output contains `re-importing`: - ``` - - Project files were missing from mem0. Re-imported successfully. - ``` -- If output contains errors or no files were processed: - ``` - - Project file import failed. Check API key and retry with: /mem0:onboard - ``` - -## Step 5: Coding categories (automatic) - -Coding categories optimized for development workflows are installed **automatically in the background on session start** — the same way project files are imported (Step 4). The user is no longer asked. This step only verifies they are in place and applies them if the background run hasn't finished yet. - -The installer is idempotent and self-caching: it compares existing categories against the proposed set, skips the API call when they already match, and skips all network calls entirely once applied for this account (re-applying only if the taxonomy changes). Safe to run anytime. - -Run it in the foreground to verify, using the plugin's venv python: - -```bash -VENV_PY="${CLAUDE_PLUGIN_DATA}/venv/bin/python3" -if [ -x "${VENV_PY}" ]; then - MEM0_DEBUG=1 "${VENV_PY}" "${CLAUDE_PLUGIN_ROOT}/scripts/auto_setup_categories.py" -else - MEM0_DEBUG=1 python3 "${CLAUDE_PLUGIN_ROOT}/scripts/auto_setup_categories.py" -fi -``` - -Parse the output: -- contains `Applied coding categories` → Print: `- Coding categories installed ( categories).` -- contains `already configured` → Print: `- Coding categories already configured.` -- error or `SDK not ready` → run the dependency installer first, then retry: -```bash -"${CLAUDE_PLUGIN_ROOT}/scripts/ensure_deps.sh" -``` -Then retry the categories script. - -## Step 6: Summary - -Print a summary: -``` -- Onboarding complete. - user_id: - project_id: (app_id) - files: found, imported - categories: - -Memory is now active for this project. Start working — mem0 will -automatically search relevant context and capture learnings. - -Run /mem0:tour to see what mem0 already knows about this project. -``` diff --git a/integrations/mem0-plugin/skills/peek/SKILL.md b/integrations/mem0-plugin/skills/peek/SKILL.md deleted file mode 100644 index ccc6b2a0c..000000000 --- a/integrations/mem0-plugin/skills/peek/SKILL.md +++ /dev/null @@ -1,52 +0,0 @@ ---- -name: peek -description: Searches memories and displays compact one-liner results, or looks up a specific memory by ID. Use for quick memory lookups, checking if a decision was recorded, resolving [mem0:id] citations, or browsing memories without full category detail. ---- - -# Mem0 Peek - -Quick search with compact output. Lighter than `/mem0:tour`. - -## Execution - -### Step 1: Parse query - -The user provides a search query: `/mem0:peek auth middleware` - -If no query provided, ask: "What should I search for?" - -**Memory ID detection:** If the query matches any of these patterns, treat it as a -direct memory ID lookup instead of a search: -- Bare hex: `^[a-f0-9]{8}$` (short ID) or `^[a-f0-9]{8}-[a-f0-9-]+$` (full UUID) -- Citation ref: `[mem0:]` — extract the hex portion - -When an ID is detected: -1. Call `get_memory()` directly (if short ID, try as prefix of full UUID) -2. If found, skip to Step 3 and display the single result -3. If not found, fall through to search using the ID as query text - -### Step 2: Search - -Run 2 parallel `search_memories` calls: - -1. Broad: `query=`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=10`, `rerank=true` -2. Targeted: `query=`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"type": "decision"}}]}`, `top_k=5`, `rerank=true` - -### Step 3: Display - -Deduplicate by ID, then show compact results: - -``` -## mem0 peek: "" ( results) - -1. [decision] Auth module uses JWT with RS256 keys (2025-05-15) [mem0:a3f8b2c1] -2. [anti_pattern] Don't use symmetric HS256 — leaked in env (2025-05-10) [mem0:7e2d9f4a] -3. [convention] All middleware in src/middleware/ (2025-05-08) [mem0:c4d5e6f7] -``` - -Format: `. [] () [mem0:]` - -If no results: -``` -No memories matching "" for project . -``` diff --git a/integrations/mem0-plugin/skills/pin/SKILL.md b/integrations/mem0-plugin/skills/pin/SKILL.md deleted file mode 100644 index 5ee20be7d..000000000 --- a/integrations/mem0-plugin/skills/pin/SKILL.md +++ /dev/null @@ -1,67 +0,0 @@ ---- -name: pin -description: Pins or unpins a memory to protect it from pruning during dream consolidation. Use when a memory is critical and must never be removed, such as architecture decisions, security constraints, or immutable team conventions. ---- - -# Mem0 Pin - -Pin a memory to mark it as high-priority and protect from pruning. - -## Execution - -### Step 1: Find the memory - -The user provides either a search query or memory ID. - -**If memory ID:** -- Call `get_memory` with the ID. - -**If search query:** -- Call `search_memories` with the query, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=5`. -- Show numbered list with content previews. -- Ask: "Which memory to pin? Enter a number." - -### Step 2: Read current content - -Call `get_memory` with the selected memory ID. Store: -- `original_text` — the memory's text content -- `original_metadata` — the existing `metadata` dict - -### Step 3: Pin it - -The MCP `update_memory` tool only accepts `memory_id`, `text`, and `source` — it -does not accept a `metadata` parameter. To pin, append a pin marker to the text: - -```python -pinned_text = "[PINNED] " + original_text if not original_text.startswith("[PINNED]") else original_text -update_memory(memory_id=, text=pinned_text) -``` - -**For new memories** (user wants to pin text that isn't stored yet): -1. Call `add_memory` with: - - `text="[PINNED] "` - - `user_id=` - - `app_id=` - - `metadata={"pinned": true, "type": "decision", "confidence": 1.0}` - - `infer=False` -2. The response contains `event_id`. Call `get_event_status(event_id=)` once to retrieve the memory ID, then confirm. - -### Step 4: Confirm - -``` -Pinned: "" -Memory ID: -``` - -Append `...` only if content exceeds 80 characters. - -### Unpin - -If the user says "unpin": -1. Call `get_memory` to read current content. -2. Remove the pin marker from the text: - ```python - unpinned_text = original_text.removeprefix("[PINNED] ") - update_memory(memory_id=, text=unpinned_text) - ``` -3. Print: `Unpinned: "..."` diff --git a/integrations/mem0-plugin/skills/policy/SKILL.md b/integrations/mem0-plugin/skills/policy/SKILL.md deleted file mode 100644 index 43bd64ed2..000000000 --- a/integrations/mem0-plugin/skills/policy/SKILL.md +++ /dev/null @@ -1,71 +0,0 @@ ---- -name: policy -description: Views or sets the project's memory-extraction policy (what Mem0 remembers vs ignores) stored in mem0.md. Use when the user says set a memory policy, custom instructions, what should mem0 remember, tell mem0 to ignore X, or asks to see/change the current instructions. ---- - -# Mem0 Policy - -The project's memory policy lives in a `## Instructions` section of `mem0.md` at -the repo root. It is version-controlled and shared by the whole team, and it maps -to Mem0's `custom_instructions` (what to extract / ignore). An optional -`## Agent Instructions` section maps to `agent_custom_instructions` (guidance for -agent-scoped memories only). - -The policy takes effect on the **next session** (mem0.md is re-parsed on -SessionStart) and is applied automatically on the plugin's memory writes. - -## Execution - -### Step 1: Determine intent from the argument - -`/mem0:policy` is invoked as `/mem0:policy [] `: - -| Argument | Action | -|---|---| -| _(none)_ or `show` | Show the current policy | -| `set ` or free text | Set/replace the `## Instructions` (custom) policy | -| `agent ` | Set/replace the `## Agent Instructions` policy | -| `clear` | Remove both policy sections | - -### Step 2: Locate mem0.md - -`mem0.md` sits at the repo root (the current working directory). If it does not -exist yet and the user is setting a policy, you will create it. - -### Step 3a: Show - -Run `python3 "$CLAUDE_PLUGIN_ROOT/scripts/parse_mem0_config.py" --key instructions .` -and `--key agent_instructions .` (or read `mem0.md` directly). Print: - -``` -Memory policy for this project (mem0.md): - Instructions: - Agent Instructions: -``` - -If `mem0.md` is absent, say there is no policy yet and offer to set one. - -### Step 3b: Set / clear - -Edit `mem0.md`, adding or replacing the target section. Keep the instruction a -short prose paragraph (what to remember, what to ignore). Example: - -```markdown -## Instructions -Remember architecture decisions, API contracts, and team conventions. -Ignore transient debugging output, stack traces, and anything resembling a secret. -``` - -- Preserve any other existing sections (`## Retention`, `## Categories`, etc.). -- If the section already exists, replace its body; otherwise append the section. -- For `clear`, delete the `## Instructions` and `## Agent Instructions` sections. - -### Step 4: Confirm - -``` -Updated memory policy in mem0.md. -Takes effect next session. Commit mem0.md to share it with your team. -``` - -Remind the user that `## Agent Instructions` only affects agent-scoped memories, -so it is a no-op unless memories are written with an `agent_id`. diff --git a/integrations/mem0-plugin/skills/remember/SKILL.md b/integrations/mem0-plugin/skills/remember/SKILL.md deleted file mode 100644 index 599c2f7a7..000000000 --- a/integrations/mem0-plugin/skills/remember/SKILL.md +++ /dev/null @@ -1,57 +0,0 @@ ---- -name: remember -description: Stores a memory verbatim from user input with appropriate type classification and metadata. Use when the user says remember this, save this, store this, note that, or explicitly asks to record a decision, preference, convention, or learning. ---- - -# Mem0 Remember - -Store a fact or learning directly into mem0. - -## Execution - -### Step 1: Extract the content - -The user provides the content as an argument: `/mem0:remember ` - -If no text was provided, ask: "What should I remember?" - -### Step 2: Classify the memory - -Based on the content, pick the best `metadata.type`: - -| Content signal | Type | -|---|---| -| "we decided...", "always use...", "never..." | `decision` | -| "X doesn't work because...", "don't try..." | `anti_pattern` | -| "I prefer...", "use X instead of Y" | `user_preference` | -| "the convention is...", "we always..." | `convention` | -| "learned that...", "figured out..." | `task_learning` | -| setup, env, tooling, config | `environmental` | -| anything else | `task_learning` | - -### Step 3: Store - -Call `add_memory` with: -- `text=""` -- `user_id=` -- `app_id=` -- `metadata={"type": "", "branch": "", "confidence": 1.0, "source": "remember_command"}` -- `infer=False` - -`infer=False` because the user stated the fact explicitly — no extraction needed. -`confidence=1.0` because the user explicitly asked to store this. - -### Step 4: Confirm - -The `add_memory` response returns `event_id` (not `memory_id`) because writes are async. -Call `get_event_status(event_id=)` once. - -- If status is `SUCCEEDED`: print the memory ID from the result. -- If status is `PENDING` or `processing`: print with the event ID as fallback. - -``` -Remembered as : "" -Memory ID: -``` - -Append `...` only if content was truncated (longer than 80 chars). diff --git a/integrations/mem0-plugin/skills/stats/SKILL.md b/integrations/mem0-plugin/skills/stats/SKILL.md deleted file mode 100644 index 5652cb58e..000000000 --- a/integrations/mem0-plugin/skills/stats/SKILL.md +++ /dev/null @@ -1,130 +0,0 @@ ---- -name: stats -description: Displays memory usage statistics for the current session and project including counts by category, age distribution, and API latency. Use when checking how many memories exist, reviewing session activity, or auditing memory distribution across categories. ---- - -# Mem0 Stats - -Show session and lifetime memory statistics. - -## Execution - -### Step 1: Gather session stats - -Run the session stats reporter: - -```bash -SCRIPT_DIR="${CLAUDE_PLUGIN_ROOT:-${CODEX_PLUGIN_ROOT:-${CURSOR_PLUGIN_ROOT:-}}}/scripts" -python3 "$SCRIPT_DIR/session_stats.py" peek 2>/dev/null || echo "{}" -``` - -The `peek` command returns JSON without clearing the stats file (unlike `report`). - -If the script returns empty or errors, note "No session data available" and continue. - -### Step 2: Fetch lifetime and session stats from API - -**Lifetime stats:** -Call `get_memories` to fetch all memories for this project: - -`filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `page_size=100` - -Group by: -1. `categories[0]` (platform-assigned) — primary grouping -2. `metadata.type` (agent-assigned) — secondary if no categories -3. `created_at` date — for age analysis - -**Category normalization:** Merge `auto_capture` and `uncategorized` into a single `uncategorized` row. These are memories where the platform didn't assign a meaningful content category. Do NOT show `auto_capture` as its own row in the table. - -**Session stats (local only):** -Session stats come from the local stats file read in Step 1. Do NOT query the API with -`run_id` or `metadata.session_id` filters — these return unreliable results because -memories are stored without `run_id` and metadata filters on `session_id` are inconsistent. - -The local stats file tracks adds and searches for the current session accurately. - -Also run a `search_memories` MCP tool call with `query="project"`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=1` to measure round-trip latency. Note the time before and after the MCP call — do NOT attempt raw HTTP calls to the API. - -### Step 3: Display - -Print a minimal dashboard. No ASCII bar charts — use a clean table layout: - -``` -## mem0 stats - -**Session** () — 3 written, 5 searches, categories: decision, convention - -**Project: my-project** — 55 memories, API: 84ms - -| Category | Count | -|----------------------|-------| -| decision | 24 | -| convention | 15 | -| anti_pattern | 6 | -| task_learning | 5 | -| user_preference | 3 | -| session_state | 2 | - -**Age** — oldest: 2026-02-15, newest: 2026-05-23 - < 7 days: 5 · 7–30d: 12 · 30–90d: 10 · > 90d: 8 - -**Identity** — user: kartik · project: my-project · branch: main -``` - -**Display rules:** -- Category table: sort by count descending, omit categories with 0 memories -- Age: single line with dot-separated buckets, computed from `created_at` -- Session line: skip if no session data available -- If only 1-2 total memories, skip the category table — just show the count -- Keep everything compact — no decorative borders or filler - -## Weekly digest mode - -When invoked with `--weekly` (e.g., `/mem0:stats --weekly`), append a weekly -activity digest after the standard stats dashboard: - -### W1: Fetch recent memories - -Call `search_memories` in parallel with time-scoped queries: -1. `query="decisions made this week"`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}, {"created_at": {"gte": "<7 days ago YYYY-MM-DD>"}}]}`, `top_k=20` -2. `query="bugs errors fixes"`, same time filter, `top_k=20` -3. `query="patterns conventions learnings"`, same time filter, `top_k=20` - -### W2: Analyze - -Merge by ID. Group into "New this week" by `categories[0]` or `metadata.type`. -Calculate: memories added last 7 days, most active categories, most active day. - -### W3: Display - -Append after the standard stats: - -``` -### This week (May 16 – May 23) - -+12 memories — most active: Wednesday (5) - -| Category | New | -|---------------|-----| -| decision | 5 | -| task_learning | 4 | -| bug_fix | 3 | - -**Highlights** -- <2-3 sentence summary of most important decisions/learnings this week> -``` - -### W4: Write digest file - -Write to `~/.mem0/weekly-digest.md` (overwrite). Append one-line to -`~/.mem0/digest-history.log`: -``` - | | + memories | top: -``` - -### W5: Empty state - -If no new memories in 7 days: -``` -No new memories in the past week. Total: memories in . -``` diff --git a/integrations/mem0-plugin/skills/switch-project/SKILL.md b/integrations/mem0-plugin/skills/switch-project/SKILL.md deleted file mode 100644 index 739e19beb..000000000 --- a/integrations/mem0-plugin/skills/switch-project/SKILL.md +++ /dev/null @@ -1,103 +0,0 @@ ---- -name: switch-project -description: Overrides the auto-detected project scope to read and write memories under a different project ID, or enables global search to access all memories across all users and projects. Use when working across multiple projects, accessing memories from another repo, enabling team-wide memory access, or when auto-detection resolves to the wrong project. ---- - -# Mem0 Switch Project - -Override the automatic project_id detection for the current directory, or enable global search mode. - -## Usage - -- `/mem0:switch-project ` — switch to a specific project scope -- `/mem0:switch-project --global` — enable global search (all memories, all users, all projects) -- `/mem0:switch-project --no-global` — disable global search and return to per-project scoping - -## Execution - -### If `--global` flag is provided: - -1. Set `global_search: true` in `~/.mem0/settings.json` using the Bash tool: - - ```bash - python3 -c " - import json, os - settings_file = os.path.expanduser('~/.mem0/settings.json') - settings = {} - if os.path.isfile(settings_file): - with open(settings_file) as f: - settings = json.load(f) - settings['global_search'] = True - with open(settings_file, 'w') as f: - json.dump(settings, f, indent=2) - print('Global search enabled') - " - ``` - -2. Print: - ``` - Global search enabled. - Searches now return all memories across all users and projects. - Writes still use the current user_id and app_id. - Restart the session for the change to take effect. - ``` - -### If `--no-global` flag is provided: - -1. Set `global_search: false` in `~/.mem0/settings.json` using the Bash tool: - - ```bash - python3 -c " - import json, os - settings_file = os.path.expanduser('~/.mem0/settings.json') - settings = {} - if os.path.isfile(settings_file): - with open(settings_file) as f: - settings = json.load(f) - settings['global_search'] = False - with open(settings_file, 'w') as f: - json.dump(settings, f, indent=2) - print('Global search disabled') - " - ``` - -2. Print: - ``` - Global search disabled. - Searches now return only memories scoped to the current project. - Restart the session for the change to take effect. - ``` - -### If a project name is provided (no flags): - -1. If no project name was given, ask: "What project_id should this directory use?" - -2. Write the mapping to `~/.mem0/project_map.json` using the Bash tool: - - ```bash - python3 -c " - import json, os - map_file = os.path.expanduser('~/.mem0/project_map.json') - mapping = {} - if os.path.isfile(map_file): - with open(map_file) as f: - mapping = json.load(f) - mapping[os.getcwd()] = '' - os.makedirs(os.path.dirname(map_file), exist_ok=True) - with open(map_file, 'w') as f: - json.dump(mapping, f, indent=2) - print(f'Mapped {os.getcwd()} -> ') - " - ``` - - (Replace `` with the user's chosen project name.) - -3. Verify by searching for existing memories: - - Call `search_memories` with `query="project"`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=1` - -4. Print: - ``` - Switched to project . - memories found for this project. - Note: This override persists across sessions for this directory. - ``` diff --git a/integrations/mem0-plugin/skills/tour/SKILL.md b/integrations/mem0-plugin/skills/tour/SKILL.md deleted file mode 100644 index 7e3bcb4aa..000000000 --- a/integrations/mem0-plugin/skills/tour/SKILL.md +++ /dev/null @@ -1,123 +0,0 @@ ---- -name: tour -description: Browses all stored memories grouped by category with full content display. Use when reviewing all project memories, exploring stored knowledge, onboarding to a project, or getting an overview of captured decisions, conventions, and learnings. ---- - -# Mem0 Project Tour - -Show the user what mem0 has stored for the current project. - -## Cross-project mode - -When invoked with `--all-projects` (e.g., `/mem0:tour --all-projects` or -`/mem0:tour --all-projects auth middleware`), search across ALL projects: - -1. Call `get_memories` with `filters={"AND": [{"user_id": ""}]}`, `page_size=200` — **no `app_id` filter**. -2. If a search query was also provided, run `search_memories` with `query=`, - `filters={"AND": [{"user_id": ""}]}`, `top_k=20` — again no `app_id`. -3. Group results by `app_id` first, then by category within each project. -4. Display: - ``` - ## ( memories) ← current - **Architecture Decisions** — - ... - - ## ( memories) - ... - - memories across projects - ``` -5. Mark the current project with `← (current)` in the heading. - -If `--all-projects` is NOT present, use the standard single-project flow below. - -## Peek mode (compact search) - -When `/mem0:tour` receives a search query argument (e.g., `/mem0:tour auth middleware`) -WITHOUT `--all-projects`, run in **peek mode** — compact one-liner results: - -1. Run 2 parallel `search_memories` calls: - - Broad: `query=`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=10`, `rerank=true` - - Targeted: `query=`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"type": "decision"}}]}`, `top_k=5`, `rerank=true` -2. Deduplicate by ID, display compact results: - ``` - ## mem0 search: "" ( results) - - 1. [decision] Auth module uses JWT with RS256 keys (2025-05-15) [mem0:a3f8b2c1] - 2. [anti_pattern] Don't use symmetric HS256 — leaked in env (2025-05-10) [mem0:7e2d9f4a] - 3. [convention] All middleware in src/middleware/ (2025-05-08) [mem0:c4d5e6f7] - ``` - Format: `. [] () [mem0:]` -3. If no results: `No memories matching "" for project .` - -If no query argument and no `--all-projects` flag, use the full tour flow below. - -## Execution - -### Step 1: Fetch ALL memories for this project - -Call `get_memories` to fetch all memories for this project: - -`filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `page_size=100` - -### Step 2: Run supplementary semantic searches - -In parallel, run these `search_memories` calls to get relevance-ranked results for key topics: - -- `query="architecture decisions design choices"`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=10`, `rerank=true` -- `query="bugs errors failures anti-patterns"`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=10`, `rerank=true` -- `query="project setup tooling conventions preferences"`, `filters={"AND": [{"user_id": ""}, {"app_id": ""}]}`, `top_k=10`, `rerank=true` - -**Do NOT filter by `metadata.type` in these calls.** The platform auto-assigns `categories` — filtering on `metadata.type` misses memories that were auto-categorized but don't have an explicit `metadata.type`. - -### Step 3: Merge and group - -Merge all results by memory ID (deduplicate). For each memory, determine its group using this priority: - -1. **Platform `categories` field** (array on each memory, auto-assigned by Mem0). Use the first category value. -2. **`metadata.type` field** (if present, set explicitly by hooks/agent). Use as fallback if no `categories`. -3. **"other"** bucket for memories with neither. - -Map category names to display names: - -| Platform category / metadata.type | Display name | -|---|---| -| `architecture decisions`, `architecture_decisions`, `decision` | Architecture Decisions | -| `anti patterns`, `anti_patterns`, `anti_pattern` | Anti-Patterns | -| `task learnings`, `task_learnings`, `task_learning` | Task Learnings | -| `coding conventions`, `coding_conventions`, `convention` | Coding Conventions | -| `user preferences`, `user_preferences`, `user_preference` | User Preferences | -| `project profile`, `project_profile` | Project Profile | -| `tooling setup`, `tooling_setup`, `environmental` | Tooling & Setup | -| `technology`, `professional_details` | Tooling & Setup | -| `session_state` | Session State | -| `compact_summary` | Compact Summaries | -| anything else | Other | - -### Step 4: Display results - -Sort groups by descending memory count. For each group that has results, print: - -``` -## ( memories) -- (score: ) -- ... -``` - -Show the **full memory text** for each entry — do NOT truncate. If a group has more than 10 entries, show top 10 by recency (or similarity score if from a search call) and note `... and more`. - -For groups with zero results, skip them entirely — don't print empty groups. - -### Step 5: Print totals - -``` - memories across categories — project: , branch: -``` - -### Step 6: Empty state - -If zero memories found for this project, print: -``` -No memories stored yet for project . -Run /mem0:onboard to import project files, or start working — mem0 captures learnings automatically. -``` diff --git a/integrations/mem0-plugin/tests/conftest.py b/integrations/mem0-plugin/tests/conftest.py deleted file mode 100644 index a170017f5..000000000 --- a/integrations/mem0-plugin/tests/conftest.py +++ /dev/null @@ -1,65 +0,0 @@ -"""Shared fixtures for mem0-plugin tests.""" - -from __future__ import annotations - -import os -import subprocess -import sys - -import pytest - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - - -@pytest.fixture(autouse=True) -def _scripts_on_path(): - """Ensure scripts/ is on sys.path so we can import _project, session_stats, etc.""" - abs_scripts = os.path.abspath(SCRIPTS_DIR) - if abs_scripts not in sys.path: - sys.path.insert(0, abs_scripts) - yield - if abs_scripts in sys.path: - sys.path.remove(abs_scripts) - - -@pytest.fixture(autouse=True) -def _isolated_home(tmp_path, monkeypatch): - """Point HOME at a tmp dir so ~/.mem0 writes never touch the real home.""" - home = tmp_path / "home" - home.mkdir() - monkeypatch.setenv("HOME", str(home)) - monkeypatch.setenv("USERPROFILE", str(home)) - monkeypatch.delenv("MEM0_PROJECT_ID", raising=False) - yield home - - -@pytest.fixture() -def tmp_git_repo(tmp_path): - """Create a temp dir with a git repo and HTTPS remote.""" - subprocess.run(["git", "init"], cwd=tmp_path, capture_output=True, check=True) - subprocess.run( - ["git", "remote", "add", "origin", "https://github.com/mem0ai/mem0.git"], - cwd=tmp_path, - capture_output=True, - check=True, - ) - return tmp_path - - -@pytest.fixture() -def tmp_git_repo_ssh(tmp_path): - """Create a temp dir with a git repo and SSH remote.""" - subprocess.run(["git", "init"], cwd=tmp_path, capture_output=True, check=True) - subprocess.run( - ["git", "remote", "add", "origin", "git@github.com:acme/cool-project.git"], - cwd=tmp_path, - capture_output=True, - check=True, - ) - return tmp_path - - -@pytest.fixture() -def tmp_no_git(tmp_path): - """Temp dir with no git repo.""" - return tmp_path diff --git a/integrations/mem0-plugin/tests/test_auto_capture.py b/integrations/mem0-plugin/tests/test_auto_capture.py deleted file mode 100644 index 3866b9503..000000000 --- a/integrations/mem0-plugin/tests/test_auto_capture.py +++ /dev/null @@ -1,142 +0,0 @@ -"""Tests for auto_capture.py transcript parsing and exchange extraction.""" - -from __future__ import annotations - -import json -import os -import sys - -import pytest - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - - -@pytest.fixture(autouse=True) -def _scripts_path(): - abs_scripts = os.path.abspath(SCRIPTS_DIR) - if abs_scripts not in sys.path: - sys.path.insert(0, abs_scripts) - yield - if abs_scripts in sys.path: - sys.path.remove(abs_scripts) - - -def _make_transcript(tmp_path, entries): - path = tmp_path / "transcript.jsonl" - lines = [] - for entry in entries: - lines.append(json.dumps(entry)) - path.write_text("\n".join(lines) + "\n") - return str(path) - - -def _msg(role, content): - return {"message": {"role": role, "content": content}} - - -class TestTailLines: - def test_reads_last_n_lines(self, tmp_path): - from auto_capture import tail_lines - - p = tmp_path / "test.txt" - p.write_text("\n".join(f"line{i}" for i in range(100)) + "\n") - result = tail_lines(str(p), 5) - assert len(result) >= 5 - assert result[-1] == "line99" - - def test_empty_file(self, tmp_path): - from auto_capture import tail_lines - - p = tmp_path / "empty.txt" - p.write_text("") - assert tail_lines(str(p), 10) == [] - - def test_nonexistent_file(self): - from auto_capture import tail_lines - - assert tail_lines("/nonexistent/path", 10) == [] - - -class TestExtractRecentExchanges: - def test_extracts_user_assistant_pairs(self): - from auto_capture import extract_recent_exchanges - - lines = [ - json.dumps(_msg("user", "What is Python used for?" * 3)), - json.dumps(_msg("assistant", "Python is used for many things." * 3)), - json.dumps(_msg("user", "Tell me about web frameworks." * 3)), - json.dumps(_msg("assistant", "Django and Flask are popular." * 3)), - ] - result = extract_recent_exchanges(lines, max_exchanges=2) - assert len(result) == 4 - assert result[0]["role"] == "user" - assert result[1]["role"] == "assistant" - - def test_skips_short_messages(self): - from auto_capture import extract_recent_exchanges - - lines = [ - json.dumps(_msg("user", "ok")), - json.dumps(_msg("assistant", "Sure, here is a detailed explanation." * 3)), - ] - result = extract_recent_exchanges(lines, max_exchanges=2) - assert len(result) == 1 - assert result[0]["role"] == "assistant" - - def test_skips_compact_summaries(self): - from auto_capture import extract_recent_exchanges - - lines = [ - json.dumps({"isCompactSummary": True, "message": {"role": "assistant", "content": "summary " * 20}}), - json.dumps(_msg("user", "This is a real user message here." * 2)), - ] - result = extract_recent_exchanges(lines, max_exchanges=2) - assert len(result) == 1 - assert result[0]["role"] == "user" - - def test_limits_to_max_exchanges(self): - from auto_capture import extract_recent_exchanges - - lines = [] - for i in range(10): - lines.append(json.dumps(_msg("user", f"Question number {i} with enough text to pass." * 2))) - lines.append(json.dumps(_msg("assistant", f"Answer number {i} with enough text to pass." * 2))) - result = extract_recent_exchanges(lines, max_exchanges=2) - assert len(result) == 4 - - def test_handles_list_content(self): - from auto_capture import extract_recent_exchanges - - lines = [ - json.dumps({"message": {"role": "user", "content": [ - {"type": "text", "text": "This is block content that is long enough." * 2}, - ]}}), - ] - result = extract_recent_exchanges(lines, max_exchanges=2) - assert len(result) == 1 - assert "block content" in result[0]["content"] - - def test_empty_lines(self): - from auto_capture import extract_recent_exchanges - - assert extract_recent_exchanges([], max_exchanges=2) == [] - - def test_truncates_long_content(self): - from auto_capture import extract_recent_exchanges - - long_text = "x" * 5000 - lines = [json.dumps(_msg("user", long_text))] - result = extract_recent_exchanges(lines, max_exchanges=1) - assert len(result) == 1 - assert len(result[0]["content"]) == 2000 - - def test_skips_tool_call_assistant_messages(self): - from auto_capture import extract_recent_exchanges - - lines = [ - json.dumps(_msg("assistant", '{"tool_calls": [{"name": "read"}]}')), - json.dumps(_msg("user", "Thanks for reading that file for me!" * 2)), - ] - result = extract_recent_exchanges(lines, max_exchanges=2) - assert len(result) == 1 - assert result[0]["role"] == "user" diff --git a/integrations/mem0-plugin/tests/test_auto_setup_categories.py b/integrations/mem0-plugin/tests/test_auto_setup_categories.py deleted file mode 100644 index 23562f232..000000000 --- a/integrations/mem0-plugin/tests/test_auto_setup_categories.py +++ /dev/null @@ -1,153 +0,0 @@ -"""Tests for auto_setup_categories.py -- the background coding-categories installer. - -Covers the pure, network-free logic: - - fingerprints (api key + category taxonomy) - - state-file gating (load / save / is_applied) - - idempotent apply via an injected fake client (no SDK, no network) - - single source of truth for the category list (shared with setup_coding_categories) -""" - -from __future__ import annotations - -import os -import sys - -SCRIPTS_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "scripts")) -if SCRIPTS_DIR not in sys.path: - sys.path.insert(0, SCRIPTS_DIR) - -import auto_setup_categories as asc # noqa: E402 -from setup_coding_categories import CODING_CATEGORIES # noqa: E402 - - -# --------------------------------------------------------------------------- # -# Fake client (dependency injection — apply_categories takes a client) # -# --------------------------------------------------------------------------- # -class _FakeProject: - def __init__(self, current): - self._current = current - self.update_calls: list = [] - - def get(self, fields=None): - return {"custom_categories": self._current} - - def update(self, custom_categories=None, **kwargs): - self.update_calls.append(custom_categories) - return {"ok": True} - - -class _FakeClient: - def __init__(self, current): - self.project = _FakeProject(current) - - -# --------------------------------------------------------------------------- # -# Source-of-truth: the taxonomy is shared, not duplicated # -# --------------------------------------------------------------------------- # -def test_uses_shared_category_list(): - """auto_setup_categories must reuse setup_coding_categories' list, not fork it.""" - assert asc.CODING_CATEGORIES is CODING_CATEGORIES - assert len(asc.CODING_CATEGORIES) == 17 - - -# --------------------------------------------------------------------------- # -# Fingerprints # -# --------------------------------------------------------------------------- # -def test_categories_fingerprint_is_stable_hex(): - fp = asc.categories_fingerprint(CODING_CATEGORIES) - assert isinstance(fp, str) - assert len(fp) == 16 - assert all(c in "0123456789abcdef" for c in fp) - # deterministic across calls - assert fp == asc.categories_fingerprint(CODING_CATEGORIES) - - -def test_categories_fingerprint_changes_when_taxonomy_changes(): - base = asc.categories_fingerprint(CODING_CATEGORIES) - changed = asc.categories_fingerprint(CODING_CATEGORIES + [{"new_one": "desc"}]) - assert base != changed - - -def test_categories_fingerprint_order_independent(): - """Reordering the same categories must yield the same fingerprint.""" - reordered = list(reversed(CODING_CATEGORIES)) - assert asc.categories_fingerprint(CODING_CATEGORIES) == asc.categories_fingerprint(reordered) - - -def test_apikey_fingerprint_is_stable_and_opaque(): - key = "m0-supersecret-abc123" - fp = asc.apikey_fingerprint(key) - assert isinstance(fp, str) - assert len(fp) == 16 - assert fp == asc.apikey_fingerprint(key) - # never leak the raw key - assert "supersecret" not in fp - assert key not in fp - - -def test_apikey_fingerprint_differs_per_key(): - assert asc.apikey_fingerprint("m0-aaa") != asc.apikey_fingerprint("m0-bbb") - - -# --------------------------------------------------------------------------- # -# State file: load / save / gating # -# --------------------------------------------------------------------------- # -def test_load_state_missing_file_returns_empty(tmp_path): - assert asc.load_state(str(tmp_path / "does_not_exist.json")) == {} - - -def test_load_state_corrupt_file_returns_empty(tmp_path): - p = tmp_path / "categories_setup.json" - p.write_text("{not valid json") - assert asc.load_state(str(p)) == {} - - -def test_save_then_load_roundtrip_creates_parent_dir(tmp_path): - p = tmp_path / "nested" / "categories_setup.json" - asc.save_state({"abc123": "deadbeef00000000"}, str(p)) - assert p.is_file() - assert asc.load_state(str(p)) == {"abc123": "deadbeef00000000"} - - -def test_is_applied_true_only_on_exact_match(): - state = {"keyfp": "catfp"} - assert asc.is_applied(state, "keyfp", "catfp") is True - # stale taxonomy fingerprint - assert asc.is_applied(state, "keyfp", "OTHER") is False - # unknown api key - assert asc.is_applied(state, "other-key", "catfp") is False - # empty state - assert asc.is_applied({}, "keyfp", "catfp") is False - - -# --------------------------------------------------------------------------- # -# apply_categories: idempotent, network-free via fake client # -# --------------------------------------------------------------------------- # -def test_apply_skips_update_when_already_matching(): - client = _FakeClient(CODING_CATEGORIES) - result = asc.apply_categories(client, CODING_CATEGORIES) - assert result == "already-configured" - assert client.project.update_calls == [] # must NOT hit the write endpoint - - -def test_apply_updates_when_no_categories_set(): - client = _FakeClient(None) - result = asc.apply_categories(client, CODING_CATEGORIES) - assert result == "applied" - assert client.project.update_calls == [CODING_CATEGORIES] - - -def test_apply_updates_when_categories_differ(): - client = _FakeClient([{"food": "consumer default"}]) - result = asc.apply_categories(client, CODING_CATEGORIES) - assert result == "applied" - assert client.project.update_calls == [CODING_CATEGORIES] - - -def test_fetch_current_categories_handles_non_dict(): - """A non-dict project response must degrade to None, not raise.""" - - class Weird: - project = type("P", (), {"get": staticmethod(lambda fields=None: "unexpected")})() - - assert asc.fetch_current_categories(Weird()) is None diff --git a/integrations/mem0-plugin/tests/test_capture_session_summary.py b/integrations/mem0-plugin/tests/test_capture_session_summary.py deleted file mode 100644 index 7ce1f1458..000000000 --- a/integrations/mem0-plugin/tests/test_capture_session_summary.py +++ /dev/null @@ -1,79 +0,0 @@ -"""Regression tests for capture_session_summary.py request body construction. - -Guards against the double-JSON-encoding bug where ``files_touched`` was stored -as a pre-serialized JSON string and then encoded a second time with the rest of -the request body — surfacing as escaped, slash-heavy blobs in the memories shown -inside Claude Code / Cursor / Codex / Antigravity (all four editors share this -script). -""" - -from __future__ import annotations - -import json - - -class _FakeResp: - status = 200 - - def __enter__(self): - return self - - def __exit__(self, *_): - return False - - -def _capture_request_body(monkeypatch): - """Patch urlopen so store_summary posts nowhere; capture the request body.""" - import capture_session_summary as css - - captured: dict = {} - - def fake_urlopen(req, timeout=0): - captured["raw"] = req.data.decode("utf-8") - captured["body"] = json.loads(captured["raw"]) - return _FakeResp() - - monkeypatch.setattr(css.urllib.request, "urlopen", fake_urlopen) - return captured, css - - -def test_files_touched_is_json_array_not_double_encoded(monkeypatch): - """files_touched must be a real JSON array, encoded exactly once.""" - captured, css = _capture_request_body(monkeypatch) - files = ["mem0/memory/main.py", "src/client/index.ts"] - - css.store_summary( - api_key="test-key", - summary_prompt="did some work", - user_id="u1", - session_id="s1", - project_id="p1", - branch="main", - files=files, - ) - - files_touched = captured["body"]["metadata"]["files_touched"] - assert isinstance(files_touched, list), ( - "files_touched must be a JSON array, not a double-encoded string; " - f"got {type(files_touched).__name__}: {files_touched!r}" - ) - assert files_touched == files - # The file paths must not appear as an escaped JSON string inside the body. - assert '\\"' not in captured["raw"] - - -def test_files_touched_omitted_when_no_files(monkeypatch): - """No files touched -> no files_touched key (unchanged behaviour).""" - captured, css = _capture_request_body(monkeypatch) - - css.store_summary( - api_key="test-key", - summary_prompt="did some work", - user_id="u1", - session_id="s1", - project_id="p1", - branch="main", - files=[], - ) - - assert "files_touched" not in captured["body"]["metadata"] diff --git a/integrations/mem0-plugin/tests/test_coding_categories.py b/integrations/mem0-plugin/tests/test_coding_categories.py deleted file mode 100644 index c4d57b62f..000000000 --- a/integrations/mem0-plugin/tests/test_coding_categories.py +++ /dev/null @@ -1,88 +0,0 @@ -"""Tests for setup_coding_categories.py -- CODING_CATEGORIES list completeness.""" - -from __future__ import annotations - -import importlib -import os -import sys - -import pytest - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - -EXPECTED_KEYS = [ - "architecture_decisions", - "anti_patterns", - "task_learnings", - "tooling_setup", - "bug_fixes", - "coding_conventions", - "user_preferences", - "dependency_decisions", - "performance_findings", - "security_constraints", - "testing_patterns", - "data_model", - "api_contracts", - "deployment_runbook", - "team_norms", - "domain_glossary", - "experiment_results", -] - - -@pytest.fixture() -def coding_categories(): - """Import CODING_CATEGORIES from setup_coding_categories, ensuring scripts/ is on path.""" - abs_scripts = os.path.abspath(SCRIPTS_DIR) - inserted = False - if abs_scripts not in sys.path: - sys.path.insert(0, abs_scripts) - inserted = True - # Force re-import in case another test already loaded a stale version - mod_name = "setup_coding_categories" - if mod_name in sys.modules: - del sys.modules[mod_name] - mod = importlib.import_module(mod_name) - yield mod.CODING_CATEGORIES - if inserted and abs_scripts in sys.path: - sys.path.remove(abs_scripts) - - -def test_total_count(coding_categories): - """CODING_CATEGORIES must contain exactly 17 entries.""" - assert len(coding_categories) == 17, ( - f"Expected 17 categories, found {len(coding_categories)}: " - f"{[list(c.keys())[0] for c in coding_categories]}" - ) - - -def test_all_expected_keys_present(coding_categories): - """Every expected category key must appear exactly once.""" - actual_keys = [list(cat.keys())[0] for cat in coding_categories] - for key in EXPECTED_KEYS: - assert key in actual_keys, f"Missing expected category key: '{key}'" - - -def test_no_duplicate_keys(coding_categories): - """No category key may appear more than once.""" - actual_keys = [list(cat.keys())[0] for cat in coding_categories] - seen = set() - duplicates = [] - for key in actual_keys: - if key in seen: - duplicates.append(key) - seen.add(key) - assert not duplicates, f"Duplicate category keys found: {duplicates}" - - -def test_each_description_is_non_empty_string(coding_categories): - """Every category must have a non-empty string description.""" - for cat in coding_categories: - assert len(cat) == 1, f"Category dict should have exactly one key, got: {cat}" - key = list(cat.keys())[0] - description = cat[key] - assert isinstance(description, str), ( - f"Category '{key}' description is not a string: {type(description)}" - ) - assert description.strip(), f"Category '{key}' has an empty description" diff --git a/integrations/mem0-plugin/tests/test_import_competing_tools.py b/integrations/mem0-plugin/tests/test_import_competing_tools.py deleted file mode 100644 index 48a76d700..000000000 --- a/integrations/mem0-plugin/tests/test_import_competing_tools.py +++ /dev/null @@ -1,355 +0,0 @@ -"""Tests for import_competing_tools.py — competing tool file importers.""" - -from __future__ import annotations - -import json -import os -import sys -from unittest import mock - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - - -# --------------------------------------------------------------------------- -# split_sections tests (unit tests on the splitter functions) -# --------------------------------------------------------------------------- - - -def test_split_by_headers_cursorrules(): - """split_by_headers correctly splits .cursorrules content on ## headers.""" - from _chunking import split_by_headers - - content = """\ -# My Cursor Rules - -Some preamble text that belongs to the first section. - -## TypeScript Conventions - -Always use strict mode. Prefer const over let. -Never use var. - -## React Patterns - -Use functional components with hooks. -Avoid class components. - -## Testing - -Write tests for all utility functions. -""" - chunks = split_by_headers(content, "## ") - assert len(chunks) == 4 # preamble + 3 sections - - # First chunk is the preamble (before any ## header) - assert "preamble text" in chunks[0] - - # Remaining chunks start with their header - assert chunks[1].startswith("## TypeScript Conventions") - assert "strict mode" in chunks[1] - - assert chunks[2].startswith("## React Patterns") - assert "functional components" in chunks[2] - - assert chunks[3].startswith("## Testing") - assert "utility functions" in chunks[3] - - -def test_split_by_headers_copilot(): - """split_by_headers correctly splits copilot-instructions.md on ## headers.""" - from _chunking import split_by_headers - - content = """\ -## Code Style - -Use 2-space indentation. Always add trailing commas. - -## Architecture - -Follow clean architecture principles. Keep business logic in domain layer. -""" - chunks = split_by_headers(content, "## ") - assert len(chunks) == 2 - assert chunks[0].startswith("## Code Style") - assert "2-space indentation" in chunks[0] - assert chunks[1].startswith("## Architecture") - assert "clean architecture" in chunks[1] - - -def test_split_by_headers_no_headers(): - """split_by_headers returns entire content as one chunk if no headers found.""" - from _chunking import split_by_headers - - content = "This file has no headers at all. Just plain text." - chunks = split_by_headers(content, "## ") - assert len(chunks) == 1 - assert "Just plain text" in chunks[0] - - -def test_split_cline_multiple_md_files(tmp_path): - """cmd_cline processes multiple .md files from memory-bank/ directory.""" - from _chunking import filter_and_truncate - - # Create a temporary memory-bank directory with .md files - mb_dir = tmp_path / "memory-bank" - mb_dir.mkdir() - - (mb_dir / "architecture.md").write_text( - "# Architecture Decisions\n\nUse microservices architecture with event sourcing." - ) - (mb_dir / "conventions.md").write_text( - "# Code Conventions\n\nAll functions must have type hints. Use black formatter." - ) - (mb_dir / "empty.md").write_text("") # empty file should be skipped - - # Read and verify we can split the files - md_files = sorted(f for f in os.listdir(str(mb_dir)) if f.endswith(".md")) - assert "architecture.md" in md_files - assert "conventions.md" in md_files - assert "empty.md" in md_files - - non_empty = [] - for filename in md_files: - filepath = os.path.join(str(mb_dir), filename) - with open(filepath) as f: - content = f.read().strip() - if content: - chunks = filter_and_truncate([content]) - non_empty.extend(chunks) - - assert len(non_empty) == 2 - assert any("microservices" in c for c in non_empty) - assert any("type hints" in c for c in non_empty) - - -def test_split_by_hr_or_headers_continue(): - """split_by_hr_or_headers correctly splits .continue/rules.md.""" - from _chunking import split_by_hr_or_headers - - content = """\ -## First Section - -Content of first section. - ---- - -## Second Section - -Content of second section. - ---- - -Third section without a header (just after HR). -""" - chunks = split_by_hr_or_headers(content) - # Should split into meaningful chunks - assert len(chunks) >= 2 - assert any("First Section" in c for c in chunks) - assert any("Second Section" in c for c in chunks) - - -def test_filter_and_truncate_skips_short(): - """filter_and_truncate skips chunks shorter than MIN_CHUNK_CHARS (50).""" - from _chunking import filter_and_truncate - - chunks = [ - "Short", # < 50 chars, should be filtered - "A" * 49, # exactly 49 chars, should be filtered - "A" * 50, # exactly 50 chars, should be kept - "A long enough chunk that definitely passes the minimum length filter.", - ] - result = filter_and_truncate(chunks) - assert len(result) == 2 - assert all(len(c) >= 50 for c in result) - - -def test_filter_and_truncate_truncates_long(): - """filter_and_truncate truncates chunks over MAX_CHUNK_CHARS (10000).""" - from _chunking import MAX_CHUNK_CHARS, filter_and_truncate - - long_chunk = "X" * (MAX_CHUNK_CHARS + 500) - result = filter_and_truncate([long_chunk]) - assert len(result) == 1 - assert len(result[0]) == MAX_CHUNK_CHARS - - -# --------------------------------------------------------------------------- -# Mock API tests -# --------------------------------------------------------------------------- - - -def _make_mock_response(status: int = 201, body: dict | None = None) -> mock.MagicMock: - """Create a mock HTTP response object.""" - if body is None: - body = {"id": "new-mem-id", "memory": "test"} - resp = mock.MagicMock() - resp.status = status - resp.read.return_value = json.dumps(body).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = mock.MagicMock(return_value=False) - return resp - - -def test_cursorrules_import_api_call(tmp_path): - """cmd_cursorrules calls the API with correct app_id (top-level), infer=False, and correct source.""" - from import_competing_tools import cmd_cursorrules - - # Create a .cursorrules file with enough content - cursorrules = tmp_path / ".cursorrules" - cursorrules.write_text( - "## TypeScript Rules\n\nAlways use strict TypeScript. Never use 'any' type. " - "Prefer interfaces over type aliases for object shapes." - ) - - captured_requests: list[dict] = [] - - def mock_urlopen(req, timeout=None): - body = json.loads(req.data.decode()) - captured_requests.append(body) - return _make_mock_response(201) - - with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-testkey"), \ - mock.patch("import_competing_tools.resolve_user_id", return_value="testuser"), \ - mock.patch("import_competing_tools.resolve_project_id", return_value="my-project"), \ - mock.patch("import_competing_tools.resolve_branch", return_value="main"), \ - mock.patch("urllib.request.urlopen", side_effect=mock_urlopen): - - original_cwd = os.getcwd() - os.chdir(str(tmp_path)) - try: - cmd_cursorrules(["--path", str(cursorrules)]) - finally: - os.chdir(original_cwd) - - assert len(captured_requests) >= 1 - - req_body = captured_requests[0] - - # app_id must be top-level (not inside metadata) - assert req_body["app_id"] == "my-project", f"Expected app_id at top level, got: {req_body}" - - # infer must be False (not "false", but the boolean False) - assert req_body["infer"] is False, f"Expected infer=False, got: {req_body['infer']}" - - # source must be cursor-import - assert req_body["metadata"]["source"] == "cursor-import", ( - f"Expected source=cursor-import, got: {req_body['metadata'].get('source')}" - ) - - # user_id must be set - assert req_body["user_id"] == "testuser" - - # messages must be a list with role/content - assert isinstance(req_body["messages"], list) - assert req_body["messages"][0]["role"] == "user" - assert len(req_body["messages"][0]["content"]) > 0 - - -def test_copilot_import_api_call(tmp_path): - """cmd_copilot calls the API with source=copilot-import.""" - from import_competing_tools import cmd_copilot - - copilot_dir = tmp_path / ".github" - copilot_dir.mkdir() - copilot_file = copilot_dir / "copilot-instructions.md" - copilot_file.write_text( - "## Code Style\n\nUse 2-space indentation. Always add trailing commas in multi-line structures. " - "Prefer const over let. Never use var in JavaScript code." - ) - - captured: list[dict] = [] - - def mock_urlopen(req, timeout=None): - captured.append(json.loads(req.data.decode())) - return _make_mock_response(201) - - with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \ - mock.patch("import_competing_tools.resolve_user_id", return_value="user1"), \ - mock.patch("import_competing_tools.resolve_project_id", return_value="proj1"), \ - mock.patch("import_competing_tools.resolve_branch", return_value="main"), \ - mock.patch("urllib.request.urlopen", side_effect=mock_urlopen): - - cmd_copilot(["--path", str(copilot_file)]) - - assert len(captured) >= 1 - assert captured[0]["metadata"]["source"] == "copilot-import" - assert captured[0]["app_id"] == "proj1" - assert captured[0]["infer"] is False - - -def test_cline_import_multiple_files(tmp_path): - """cmd_cline imports one memory per non-empty .md file.""" - from import_competing_tools import cmd_cline - - mb = tmp_path / "memory-bank" - mb.mkdir() - (mb / "arch.md").write_text( - "Architecture: microservices with event-sourcing. Each service owns its database. " - "Communication via message bus only. No direct service-to-service HTTP calls." - ) - (mb / "style.md").write_text( - "Code style: PEP 8 for Python. Black formatter. isort for imports. " - "Line length 120. Type hints required on all public functions and methods." - ) - (mb / "empty.md").write_text("") - - captured: list[dict] = [] - - def mock_urlopen(req, timeout=None): - captured.append(json.loads(req.data.decode())) - return _make_mock_response(201) - - with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \ - mock.patch("import_competing_tools.resolve_user_id", return_value="user1"), \ - mock.patch("import_competing_tools.resolve_project_id", return_value="proj1"), \ - mock.patch("import_competing_tools.resolve_branch", return_value="main"), \ - mock.patch("urllib.request.urlopen", side_effect=mock_urlopen): - - cmd_cline(["--path", str(mb)]) - - # Should have exactly 2 imports (empty.md skipped) - assert len(captured) == 2 - sources = {r["metadata"]["source"] for r in captured} - assert sources == {"cline-import"} - for r in captured: - assert r["infer"] is False - assert r["app_id"] == "proj1" - - -def test_no_api_key_does_not_call_api(tmp_path): - """When no API key is set, no HTTP call is made.""" - from import_competing_tools import cmd_cursorrules - - cursorrules = tmp_path / ".cursorrules" - cursorrules.write_text("## Rules\n\n" + "x" * 100) - - with mock.patch("import_competing_tools.resolve_api_key", return_value=""), \ - mock.patch("urllib.request.urlopen") as mock_url: - - cmd_cursorrules(["--path", str(cursorrules)]) - - mock_url.assert_not_called() - - -def test_missing_file_does_not_call_api(tmp_path): - """When the source file doesn't exist, no HTTP call is made.""" - from import_competing_tools import cmd_cursorrules - - with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \ - mock.patch("urllib.request.urlopen") as mock_url: - - cmd_cursorrules(["--path", str(tmp_path / "nonexistent.cursorrules")]) - - mock_url.assert_not_called() - - -def test_main_unknown_subcommand_exits_zero(): - """Calling main() with an unknown subcommand exits 0.""" - import subprocess - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "import_competing_tools.py"), "unknown"], - capture_output=True, - text=True, - env={**os.environ, "MEM0_API_KEY": ""}, - ) - assert result.returncode == 0 diff --git a/integrations/mem0-plugin/tests/test_instructions.py b/integrations/mem0-plugin/tests/test_instructions.py deleted file mode 100644 index cbbe41005..000000000 --- a/integrations/mem0-plugin/tests/test_instructions.py +++ /dev/null @@ -1,126 +0,0 @@ -"""Tests for the mem0.md extraction-policy feature. - -Covers the parser (`parse_section_text` + the Instructions sections in -`load_full_config`) and the `_instructions.load_instructions` helper that the -hook writers use to attach `custom_instructions` / `agent_custom_instructions` -to a memory write. -""" - -from __future__ import annotations - -MEM0_MD = """\ -# mem0.md - -## Categories -- architecture_decisions - -## Instructions -Remember architecture decisions and conventions. -Ignore transient debug output and secrets (issue #123 refs are fine). - -## Agent Instructions -For agent memories, focus on tools and task outcomes. -""" - - -# --------------------------------------------------------------------------- -# parse_section_text -# --------------------------------------------------------------------------- - - -def test_parse_section_text_collapses_prose(): - from parse_mem0_config import parse_section_text - - text = parse_section_text(MEM0_MD, "Instructions") - assert text == ( - "Remember architecture decisions and conventions. " - "Ignore transient debug output and secrets (issue #123 refs are fine)." - ) - - -def test_parse_section_text_preserves_inline_hash(): - """Inline '#' (e.g. issue refs) is kept; only full-line comments are dropped.""" - from parse_mem0_config import parse_section_text - - assert "#123" in parse_section_text(MEM0_MD, "Instructions") - - -def test_instructions_and_agent_instructions_are_distinct(): - """'## Instructions' must not swallow '## Agent Instructions'.""" - from parse_mem0_config import parse_section_text - - assert parse_section_text(MEM0_MD, "Instructions").startswith("Remember architecture") - assert parse_section_text(MEM0_MD, "Agent Instructions") == ( - "For agent memories, focus on tools and task outcomes." - ) - - -def test_parse_section_text_missing_returns_empty(): - from parse_mem0_config import parse_section_text - - assert parse_section_text("## Retention\nx: 1d\n", "Instructions") == "" - - -def test_load_full_config_includes_instructions(tmp_path): - from parse_mem0_config import load_full_config - - (tmp_path / "mem0.md").write_text(MEM0_MD) - config = load_full_config(str(tmp_path)) - assert config["instructions"].startswith("Remember architecture") - assert config["agent_instructions"].startswith("For agent memories") - - -# --------------------------------------------------------------------------- -# load_instructions (the helper the hook writers call) -# --------------------------------------------------------------------------- - - -def test_load_instructions_maps_to_api_field_names(tmp_path): - from _instructions import load_instructions - - (tmp_path / "mem0.md").write_text(MEM0_MD) - out = load_instructions(str(tmp_path)) - assert out == { - "custom_instructions": ( - "Remember architecture decisions and conventions. " - "Ignore transient debug output and secrets (issue #123 refs are fine)." - ), - "agent_custom_instructions": "For agent memories, focus on tools and task outcomes.", - } - - -def test_load_instructions_no_config_is_empty(tmp_path): - """No mem0.md -> empty dict, so a write body gains nothing.""" - from _instructions import load_instructions - - assert load_instructions(str(tmp_path)) == {} - - -def test_load_instructions_only_custom(tmp_path): - """A project with only '## Instructions' omits the agent key entirely.""" - from _instructions import load_instructions - - (tmp_path / "mem0.md").write_text("## Instructions\nRemember decisions.\n") - out = load_instructions(str(tmp_path)) - assert out == {"custom_instructions": "Remember decisions."} - - -def test_load_instructions_defaults_cwd_to_env(tmp_path, monkeypatch): - """With no arg, cwd falls back to MEM0_CWD.""" - from _instructions import load_instructions - - (tmp_path / "mem0.md").write_text("## Instructions\nRemember decisions.\n") - monkeypatch.setenv("MEM0_CWD", str(tmp_path)) - assert load_instructions() == {"custom_instructions": "Remember decisions."} - - -def test_load_instructions_body_merge_shape(tmp_path): - """The result merges cleanly into an add body without clobbering other keys.""" - from _instructions import load_instructions - - (tmp_path / "mem0.md").write_text("## Instructions\nRemember decisions.\n") - body = {"messages": [], "user_id": "u", "infer": True} - body.update(load_instructions(str(tmp_path))) - assert body["user_id"] == "u" and body["infer"] is True - assert body["custom_instructions"] == "Remember decisions." - assert "agent_custom_instructions" not in body diff --git a/integrations/mem0-plugin/tests/test_kimi_manifests.py b/integrations/mem0-plugin/tests/test_kimi_manifests.py deleted file mode 100644 index 942f2ced6..000000000 --- a/integrations/mem0-plugin/tests/test_kimi_manifests.py +++ /dev/null @@ -1,80 +0,0 @@ -"""Validate the Kimi Code plugin manifests and hook wiring. - -Kimi Code loads a plugin from `.kimi-plugin/plugin.json`. This test locks down -the structural contract so a stray edit cannot silently break the plugin: - - - both manifests parse and carry the required Kimi fields - - the MCP server uses Kimi's supported auth (transport/url/bearerTokenEnvVar) - - every hook command routes through kimi_hook_shim.sh and targets a script - that actually exists in scripts/ - - the sessionStart skill and skills/ path resolve on disk -""" - -import json -import os - -PLUGIN_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) -REPO_ROOT = os.path.dirname(os.path.dirname(PLUGIN_DIR)) - -SUBDIR_MANIFEST = os.path.join(PLUGIN_DIR, ".kimi-plugin", "plugin.json") -ROOT_MARKETPLACE = os.path.join(REPO_ROOT, ".kimi-plugin", "marketplace.json") -SHIM = os.path.join(PLUGIN_DIR, "scripts", "kimi_hook_shim.sh") - - -def _load(path): - with open(path) as fh: - return json.load(fh) - - -def test_manifest_parses_and_has_required_fields(): - m = _load(SUBDIR_MANIFEST) - assert m["name"] == "mem0" - for field in ("version", "description", "skills", "mcpServers", "interface", "hooks"): - assert field in m, f"manifest missing {field}" - - -def test_mcp_uses_kimi_supported_auth(): - srv = _load(SUBDIR_MANIFEST)["mcpServers"]["mem0"] - assert srv["transport"] == "http" - assert srv["url"] == "https://mcp.mem0.ai/mcp/" # trailing slash avoids the 307 redirect - # Kimi reads the named env var and sends Authorization: Bearer . - assert srv["bearerTokenEnvVar"] == "MEM0_API_KEY" - assert "headers" not in srv # Kimi does not interpolate ${env:...} inside headers - - -def test_sessionstart_skill_and_skills_dir_resolve(): - m = _load(SUBDIR_MANIFEST) - assert m["skills"] == "./skills/" - assert os.path.isdir(os.path.join(PLUGIN_DIR, "skills")) - start_skill = m["sessionStart"]["skill"] - assert os.path.isdir(os.path.join(PLUGIN_DIR, "skills", start_skill)) - - -def test_shim_exists_and_is_executable(): - assert os.path.isfile(SHIM) - assert os.access(SHIM, os.X_OK), "kimi_hook_shim.sh must be executable" - - -def test_every_hook_routes_through_shim_to_a_real_script(): - hooks = _load(SUBDIR_MANIFEST)["hooks"] - assert hooks, "no hooks declared" - valid_events = { - "SessionStart", "UserPromptSubmit", "PreToolUse", - "PostToolUse", "Stop", "PreCompact", - } - for hook in hooks: - assert hook["event"] in valid_events, f"unknown event {hook['event']}" - cmd = hook["command"] - assert "kimi_hook_shim.sh" in cmd, f"hook must route through the shim: {cmd}" - assert "$KIMI_PLUGIN_ROOT/scripts/kimi_hook_shim.sh" in cmd - target = cmd.rsplit('"', 1)[-1].strip() # the script name after the shim - assert target.endswith(".sh") - assert os.path.isfile(os.path.join(PLUGIN_DIR, "scripts", target)), \ - f"hook target script missing: {target}" - - -def test_root_marketplace_catalog(): - cat = _load(ROOT_MARKETPLACE) - entry = cat["plugins"][0] - assert entry["id"] == "mem0" # Kimi marketplace entries key on id - assert entry["source"].startswith("https://github.com/mem0ai/mem0") diff --git a/integrations/mem0-plugin/tests/test_load_settings.py b/integrations/mem0-plugin/tests/test_load_settings.py deleted file mode 100644 index 0f3216256..000000000 --- a/integrations/mem0-plugin/tests/test_load_settings.py +++ /dev/null @@ -1,86 +0,0 @@ -"""Tests for scripts/load_settings.py.""" - -from __future__ import annotations - -import json -import os -import subprocess -import sys - -import pytest - - -@pytest.fixture() -def settings(tmp_path, monkeypatch): - import load_settings - - path = tmp_path / ".mem0" / "settings.json" - monkeypatch.setattr(load_settings, "SETTINGS_PATH", path) - return load_settings - - -def test_creates_file_when_absent(settings): - assert settings.create_default_settings() is True - assert settings.SETTINGS_PATH.exists() - assert json.loads(settings.SETTINGS_PATH.read_text()) == settings.DEFAULTS - - -def test_does_not_recreate_or_overwrite_existing(settings): - settings.SETTINGS_PATH.parent.mkdir(parents=True) - settings.SETTINGS_PATH.write_text('{"search_limit": 42}') - - assert settings.create_default_settings() is False - assert json.loads(settings.SETTINGS_PATH.read_text()) == {"search_limit": 42} - - -def test_user_values_override_defaults(settings): - settings.SETTINGS_PATH.parent.mkdir(parents=True) - settings.SETTINGS_PATH.write_text('{"search_limit": 3, "auto_save": false}') - - loaded = settings.load_settings() - assert loaded["search_limit"] == 3 - assert loaded["auto_save"] is False - assert loaded["global_search"] == settings.DEFAULTS["global_search"] - - -def test_unknown_keys_are_dropped_but_reported(settings): - settings.SETTINGS_PATH.parent.mkdir(parents=True) - settings.SETTINGS_PATH.write_text('{"skip_tools": ["Read"], "output_style": "compact"}') - - assert "skip_tools" not in settings.load_settings() - assert settings.unknown_keys() == ["output_style", "skip_tools"] - - -def test_unknown_keys_empty_for_clean_file(settings): - settings.create_default_settings() - assert settings.unknown_keys() == [] - - -@pytest.mark.parametrize("body", ["{not json", '["a", "list"]']) -def test_malformed_file_falls_back_to_defaults(settings, body): - settings.SETTINGS_PATH.parent.mkdir(parents=True) - settings.SETTINGS_PATH.write_text(body) - - assert settings.load_settings() == settings.DEFAULTS - assert settings.unknown_keys() == [] - - -def test_init_announces_creation_only_once(_isolated_home): - import load_settings - - home = _isolated_home - - def run_init(): - return subprocess.run( - [sys.executable, load_settings.__file__, "init"], - capture_output=True, - text=True, - check=True, - env={**os.environ, "HOME": str(home)}, - ).stdout - - first = run_init() - second = run_init() - - assert "Created" in first - assert "Created" not in second diff --git a/integrations/mem0-plugin/tests/test_message_roles.py b/integrations/mem0-plugin/tests/test_message_roles.py deleted file mode 100644 index 0a36ca66e..000000000 --- a/integrations/mem0-plugin/tests/test_message_roles.py +++ /dev/null @@ -1,144 +0,0 @@ -"""Regression tests: assistant-authored text must never be posted as role="user". - -The Stop hook (capture_session_summary) and the post-compact hook -(capture_compact_summary) both ship *model-authored* prose to -POST /v3/memories/add/. Mem0's fact extractor renders each message as -"{role}: {content}" and is instructed to extract "facts and preferences about -the user" — so role is the only signal separating what the human said from what -Claude said. - -Posting Claude's own words under role="user" made the extractor read Claude's -first-person prose ("I recommend pgvector", "I found the bug in auth.py") as the -*human's* statements and store them under their user_id. The Stop hook fires on -every assistant turn, so this corrupted memory on nearly every message. -""" - -from __future__ import annotations - -import json - - -class _FakeResp: - status = 200 - - def __enter__(self): - return self - - def __exit__(self, *_): - return False - - -def _capture(monkeypatch, module): - """Patch urlopen so store_summary posts nowhere; capture the request body.""" - captured: dict = {} - - def fake_urlopen(req, timeout=0): - captured["body"] = json.loads(req.data.decode("utf-8")) - return _FakeResp() - - monkeypatch.setattr(module.urllib.request, "urlopen", fake_urlopen) - return captured - - -# Claude's own voice — first-person prose that must never be attributed to the human. -ASSISTANT_PROSE = ( - "I traced the root cause to auth.py and I recommend we switch to pgvector " - "for the vector store. I'll refactor the session handler next." -) - - -def test_session_summary_posts_assistant_prose_as_assistant(monkeypatch): - """Stop hook: the last assistant message must be tagged role="assistant".""" - import capture_session_summary as css - - captured = _capture(monkeypatch, css) - - css.store_summary( - api_key="test-key", - summary_prompt=css.build_summary_prompt(ASSISTANT_PROSE, []), - user_id="u1", - session_id="s1", - project_id="p1", - branch="main", - files=[], - ) - - messages = captured["body"]["messages"] - for msg in messages: - if ASSISTANT_PROSE in msg["content"]: - assert msg["role"] == "assistant", ( - "Claude's own words were posted as role='user' — mem0 will extract " - "them as facts about the human. Got role=%r" % msg["role"] - ) - break - else: - raise AssertionError("assistant prose never made it into the payload") - - -def test_compact_summary_posts_assistant_prose_as_assistant(monkeypatch): - """Post-compact hook: the compact summary is model-authored, not user-authored.""" - import capture_compact_summary as ccs - - captured = _capture(monkeypatch, ccs) - - ccs.store_summary( - api_key="test-key", - summary=ASSISTANT_PROSE, - user_id="u1", - session_id="s1", - project_id="p1", - branch="main", - ) - - messages = captured["body"]["messages"] - for msg in messages: - if ASSISTANT_PROSE in msg["content"]: - assert msg["role"] == "assistant", ( - "Compact summary (written by Claude) was posted as role='user'. Got role=%r" % msg["role"] - ) - break - else: - raise AssertionError("assistant prose never made it into the payload") - - -def test_no_user_role_message_carries_assistant_prose(monkeypatch): - """Belt and braces: no user-role message may contain the assistant's words.""" - import capture_session_summary as css - - captured = _capture(monkeypatch, css) - - css.store_summary( - api_key="test-key", - summary_prompt=css.build_summary_prompt(ASSISTANT_PROSE, ["auth.py"]), - user_id="u1", - session_id="s1", - project_id="p1", - branch="main", - files=["auth.py"], - ) - - for msg in captured["body"]["messages"]: - if msg["role"] == "user": - assert ASSISTANT_PROSE not in msg["content"], ( - "A user-role message carries Claude's prose — this is the misattribution bug." - ) - - -def test_auto_capture_preserves_real_roles(): - """auto_capture is the reference: it must pass roles through untouched.""" - import auto_capture - - lines = [ - json.dumps({"type": "user", "message": {"role": "user", "content": "why is the build failing on main?"}}), - json.dumps( - { - "type": "assistant", - "message": {"role": "assistant", "content": [{"type": "text", "text": ASSISTANT_PROSE}]}, - } - ), - ] - - messages = auto_capture.extract_recent_exchanges(lines) - - assert [m["role"] for m in messages] == ["user", "assistant"] - assert ASSISTANT_PROSE in messages[1]["content"] diff --git a/integrations/mem0-plugin/tests/test_parse_export_file.py b/integrations/mem0-plugin/tests/test_parse_export_file.py deleted file mode 100644 index 66dccbd1d..000000000 --- a/integrations/mem0-plugin/tests/test_parse_export_file.py +++ /dev/null @@ -1,298 +0,0 @@ -"""Tests for parse_export_file.py — mem0 export file parser.""" - -from __future__ import annotations - -import json -import os -import subprocess -import sys - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - - -# --------------------------------------------------------------------------- -# Direct function tests -# --------------------------------------------------------------------------- - - -def test_parse_blocks_single_valid_block(): - """parse_blocks returns one record for a single valid block.""" - from parse_export_file import parse_blocks - - content = """\ ---- -id: abc123 -created_at: 2024-01-15T10:00:00Z -type: task_learnings -confidence: 0.85 -branch: main -files: src/foo.py, src/bar.py -categories: coding_conventions, task_learnings ---- -Always use context managers when opening files. -""" - records = parse_blocks(content) - assert len(records) == 1 - r = records[0] - assert r["id"] == "abc123" - assert r["type"] == "task_learnings" - assert r["confidence"] == "0.85" - assert r["branch"] == "main" - assert r["files"] == ["src/foo.py", "src/bar.py"] - assert r["categories"] == ["coding_conventions", "task_learnings"] - assert "Always use context managers" in r["content"] - - -def test_parse_blocks_multiple_blocks(): - """parse_blocks returns the correct number of records for multiple blocks.""" - from parse_export_file import parse_blocks - - content = """\ ---- -id: mem001 -type: architecture_decisions -confidence: 0.9 -branch: main -files: -categories: architecture_decisions ---- -Use hexagonal architecture for the core domain. - ---- -id: mem002 -type: anti_patterns -confidence: 0.75 -branch: feat/refactor -files: src/legacy.py -categories: anti_patterns ---- -Avoid direct database calls from view layer. - ---- -id: mem003 -type: coding_conventions -confidence: 0.8 -branch: main -files: src/utils.py, src/helpers.py -categories: ---- -Use snake_case for all Python identifiers. -""" - records = parse_blocks(content) - assert len(records) == 3 - - assert records[0]["id"] == "mem001" - assert records[0]["categories"] == ["architecture_decisions"] - assert "hexagonal architecture" in records[0]["content"] - - assert records[1]["id"] == "mem002" - assert records[1]["files"] == ["src/legacy.py"] - assert "direct database calls" in records[1]["content"] - - assert records[2]["id"] == "mem003" - assert records[2]["files"] == ["src/utils.py", "src/helpers.py"] - assert records[2]["categories"] == [] - assert "snake_case" in records[2]["content"] - - -def test_parse_blocks_missing_optional_fields(): - """parse_blocks uses defaults when optional fields are absent.""" - from parse_export_file import parse_blocks - - # confidence, branch, files, categories all absent - content = """\ ---- -id: xyz789 -type: task_learnings ---- -Run tests before committing. -""" - records = parse_blocks(content) - assert len(records) == 1 - r = records[0] - assert r["id"] == "xyz789" - assert r["confidence"] == "" # default empty string - assert r["branch"] == "" # default empty string - assert r["files"] == [] # default empty list - assert r["categories"] == [] # default empty list - assert "Run tests" in r["content"] - - -def test_parse_blocks_filters_empty_content(): - """parse_blocks skips blocks whose content is empty or whitespace-only.""" - from parse_export_file import parse_blocks - - content = """\ ---- -id: empty1 -type: task_learnings ---- - ---- -id: real1 -type: task_learnings ---- -This block has real content. - ---- -id: empty2 -type: coding_conventions ---- - -""" - records = parse_blocks(content) - # Only the block with actual content should be returned - assert len(records) == 1 - assert records[0]["id"] == "real1" - assert "real content" in records[0]["content"] - - -def test_parse_blocks_round_trip(): - """Content formatted by export matches what parse_blocks expects.""" - from parse_export_file import parse_blocks - - # Simulate the exact format produced by the export skill - memory_id = "test-id-001" - created_at = "2024-06-01T12:00:00Z" - mem_type = "architecture_decisions" - confidence = "0.92" - branch = "feat/new-feature" - files = ["src/main.py", "tests/test_main.py"] - categories = ["architecture_decisions", "coding_conventions"] - memory_content = "Use dependency injection for all service classes." - - # Format exactly as the export skill would - block = ( - "---\n" - f"id: {memory_id}\n" - f"created_at: {created_at}\n" - f"type: {mem_type}\n" - f"confidence: {confidence}\n" - f"branch: {branch}\n" - f"files: {', '.join(files)}\n" - f"categories: {', '.join(categories)}\n" - "---\n" - f"{memory_content}\n" - "\n" - ) - - records = parse_blocks(block) - assert len(records) == 1 - r = records[0] - assert r["id"] == memory_id - assert r["type"] == mem_type - assert r["confidence"] == confidence - assert r["branch"] == branch - assert r["files"] == files - assert r["categories"] == categories - assert r["content"] == memory_content - - -def test_parse_blocks_multiline_content(): - """parse_blocks correctly captures multi-line memory content.""" - from parse_export_file import parse_blocks - - content = """\ ---- -id: multi001 -type: task_learnings ---- -Line one of the memory. -Line two of the memory. - -Line four after blank line. -""" - records = parse_blocks(content) - assert len(records) == 1 - assert "Line one" in records[0]["content"] - assert "Line two" in records[0]["content"] - assert "Line four" in records[0]["content"] - - -def test_parse_blocks_empty_input(): - """parse_blocks returns empty list for empty input.""" - from parse_export_file import parse_blocks - - assert parse_blocks("") == [] - assert parse_blocks(" \n ") == [] - - -def test_parse_blocks_no_blocks(): - """parse_blocks returns empty list for content without any --- delimiters.""" - from parse_export_file import parse_blocks - - assert parse_blocks("Just some text without any delimiters.") == [] - - -def test_parse_blocks_value_with_colon(): - """parse_blocks handles values that themselves contain colons.""" - from parse_export_file import parse_blocks - - content = """\ ---- -id: colon-test -type: task_learnings -created_at: 2024-01-01T10:00:00Z ---- -Timestamp values contain colons and should parse correctly. -""" - records = parse_blocks(content) - assert len(records) == 1 - assert records[0]["id"] == "colon-test" - # created_at field should be captured (it's in the record if present) - assert "2024-01-01T10:00:00Z" in records[0].get("created_at", "") - - -# --------------------------------------------------------------------------- -# CLI / subprocess tests -# --------------------------------------------------------------------------- - - -def test_main_cli_outputs_json(tmp_path): - """Running parse_export_file.py as a script outputs valid JSON.""" - export_file = tmp_path / "mem0-export-test.md" - export_file.write_text("""\ ---- -id: cli-test-001 -type: task_learnings -confidence: 0.8 -branch: main -files: -categories: task_learnings ---- -Prefer composition over inheritance. -""") - - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py"), str(export_file)], - capture_output=True, - text=True, - ) - assert result.returncode == 0 - records = json.loads(result.stdout) - assert isinstance(records, list) - assert len(records) == 1 - assert records[0]["id"] == "cli-test-001" - - -def test_main_cli_no_args_exits_zero(): - """Running parse_export_file.py with no arguments exits 0 and prints [].""" - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py")], - capture_output=True, - text=True, - ) - assert result.returncode == 0 - assert result.stdout.strip() == "[]" - - -def test_main_cli_missing_file_exits_zero(tmp_path): - """Running parse_export_file.py with a missing file exits 0 and prints [].""" - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py"), - str(tmp_path / "nonexistent.md")], - capture_output=True, - text=True, - ) - assert result.returncode == 0 - assert result.stdout.strip() == "[]" diff --git a/integrations/mem0-plugin/tests/test_parse_mem0_config.py b/integrations/mem0-plugin/tests/test_parse_mem0_config.py deleted file mode 100644 index 4ba261f6d..000000000 --- a/integrations/mem0-plugin/tests/test_parse_mem0_config.py +++ /dev/null @@ -1,439 +0,0 @@ -"""Tests for parse_mem0_config.py — mem0.md retention policy parser.""" - -from __future__ import annotations - -import json -import os -import subprocess -import sys - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - - -# --------------------------------------------------------------------------- -# parse_retention — unit tests -# --------------------------------------------------------------------------- - - -def test_parse_retention_valid_section(): - """parse_retention extracts day-count policies correctly.""" - from parse_mem0_config import parse_retention - - content = """\ -# Project Config - -## Retention - -session_state: 90d -compact_summary: 60d -decision: 180d -""" - result = parse_retention(content) - assert result == { - "session_state": 90, - "compact_summary": 60, - "decision": 180, - } - - -def test_parse_retention_forever_returns_none(): - """parse_retention maps 'forever' to None.""" - from parse_mem0_config import parse_retention - - content = """\ -## Retention - -user_preference: forever -anti_pattern: forever -session_state: 30d -""" - result = parse_retention(content) - assert result["user_preference"] is None - assert result["anti_pattern"] is None - assert result["session_state"] == 30 - - -def test_parse_retention_no_section_returns_empty(): - """parse_retention returns {} when there is no ## Retention heading.""" - from parse_mem0_config import parse_retention - - content = """\ -# Project Config - -## Some Other Section - -key: value -""" - result = parse_retention(content) - assert result == {} - - -def test_parse_retention_stops_at_next_heading(): - """parse_retention stops reading at the next ## heading.""" - from parse_mem0_config import parse_retention - - content = """\ -## Retention - -session_state: 7d - -## Other Section - -other_key: 999d -""" - result = parse_retention(content) - assert "session_state" in result - assert "other_key" not in result - - -def test_parse_retention_malformed_lines_skipped(): - """Malformed lines (no colon, bad day format) are silently ignored.""" - from parse_mem0_config import parse_retention - - content = """\ -## Retention - -session_state: 90d -bad_line_no_colon -another: badvalue -decision: 30d -""" - result = parse_retention(content) - assert result == {"session_state": 90, "decision": 30} - - -def test_parse_retention_comments_ignored(): - """Inline # comments are stripped before parsing.""" - from parse_mem0_config import parse_retention - - content = """\ -## Retention - -session_state: 90d # rolling 90-day window -user_preference: forever # never prune preferences -""" - result = parse_retention(content) - assert result["session_state"] == 90 - assert result["user_preference"] is None - - -def test_parse_retention_case_insensitive_heading(): - """## retention (lowercase) is matched the same as ## Retention.""" - from parse_mem0_config import parse_retention - - content = """\ -## retention - -session_state: 14d -""" - result = parse_retention(content) - assert result == {"session_state": 14} - - -def test_parse_retention_empty_section_returns_empty(): - """A ## Retention section with no valid lines returns {}.""" - from parse_mem0_config import parse_retention - - content = """\ -## Retention - -# only comments here - -## Next Section -""" - result = parse_retention(content) - assert result == {} - - -# --------------------------------------------------------------------------- -# load_retention_policies — integration tests with tmp files -# --------------------------------------------------------------------------- - - -def test_load_retention_policies_with_tmp_file(tmp_path): - """load_retention_policies reads a real mem0.md from disk.""" - from parse_mem0_config import load_retention_policies - - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text( - """\ -# My Project - -## Retention - -session_state: 90d -compact_summary: 60d -decision: forever -""", - encoding="utf-8", - ) - - result = load_retention_policies(str(tmp_path)) - assert result == { - "session_state": 90, - "compact_summary": 60, - "decision": None, - } - - -def test_load_retention_policies_no_mem0_md_returns_empty(tmp_path): - """load_retention_policies returns {} when no mem0.md exists.""" - from parse_mem0_config import load_retention_policies - - result = load_retention_policies(str(tmp_path)) - assert result == {} - - -def test_load_retention_policies_no_retention_section_returns_empty(tmp_path): - """load_retention_policies returns {} when mem0.md has no ## Retention.""" - from parse_mem0_config import load_retention_policies - - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text( - """\ -# My Project - -Some general project notes here. -No retention section. -""", - encoding="utf-8", - ) - - result = load_retention_policies(str(tmp_path)) - assert result == {} - - -def test_load_retention_policies_defaults_to_cwd(tmp_path, monkeypatch): - """load_retention_policies uses os.getcwd() when cwd is None.""" - from parse_mem0_config import load_retention_policies - - monkeypatch.chdir(tmp_path) - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text("## Retention\nsession_state: 45d\n", encoding="utf-8") - - result = load_retention_policies() # no cwd arg - assert result == {"session_state": 45} - - -# --------------------------------------------------------------------------- -# CLI / main() — subprocess test -# --------------------------------------------------------------------------- - - -def test_cli_main_prints_json(tmp_path): - """CLI: python parse_mem0_config.py prints valid JSON.""" - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text( - "## Retention\nsession_state: 90d\nuser_preference: forever\n", - encoding="utf-8", - ) - - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "parse_mem0_config.py"), str(tmp_path)], - capture_output=True, - text=True, - ) - assert result.returncode == 0 - data = json.loads(result.stdout) - assert data["session_state"] == 90 - assert data["user_preference"] is None - - -def test_cli_main_no_file_prints_empty_json(tmp_path): - """CLI: prints '{}' when no mem0.md exists.""" - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "parse_mem0_config.py"), str(tmp_path)], - capture_output=True, - text=True, - ) - assert result.returncode == 0 - assert json.loads(result.stdout) == {} - - -# --------------------------------------------------------------------------- -# parse_section_kv — unit tests -# --------------------------------------------------------------------------- - - -def test_parse_section_kv_basic(): - """parse_section_kv extracts key-value pairs from a named section.""" - from parse_mem0_config import parse_section_kv - - content = """\ -## Search - -default_limit: 10 -boost_recency: true -""" - result = parse_section_kv(content, "Search") - assert result == {"default_limit": "10", "boost_recency": "true"} - - -def test_parse_section_kv_missing_section(): - """parse_section_kv returns {} when section doesn't exist.""" - from parse_mem0_config import parse_section_kv - - result = parse_section_kv("## Other\nfoo: bar\n", "Search") - assert result == {} - - -def test_parse_section_kv_stops_at_next_heading(): - """parse_section_kv stops at the next ## heading.""" - from parse_mem0_config import parse_section_kv - - content = """\ -## Identity - -user_id: kartik -project_id: mem0 - -## Other - -ignored: yes -""" - result = parse_section_kv(content, "Identity") - assert result == {"user_id": "kartik", "project_id": "mem0"} - assert "ignored" not in result - - -# --------------------------------------------------------------------------- -# parse_section_list — unit tests -# --------------------------------------------------------------------------- - - -def test_parse_section_list_basic(): - """parse_section_list extracts list items from a named section.""" - from parse_mem0_config import parse_section_list - - content = """\ -## Categories - -- architecture_decisions -- bug_fixes -- coding_conventions -""" - result = parse_section_list(content, "Categories") - assert result == ["architecture_decisions", "bug_fixes", "coding_conventions"] - - -def test_parse_section_list_bare_lines(): - """parse_section_list works with bare lines (no bullet prefix).""" - from parse_mem0_config import parse_section_list - - content = """\ -## Categories - -architecture_decisions -bug_fixes -""" - result = parse_section_list(content, "Categories") - assert result == ["architecture_decisions", "bug_fixes"] - - -def test_parse_section_list_missing_section(): - """parse_section_list returns [] when section doesn't exist.""" - from parse_mem0_config import parse_section_list - - result = parse_section_list("## Other\n- foo\n", "Categories") - assert result == [] - - -# --------------------------------------------------------------------------- -# load_full_config — integration tests -# --------------------------------------------------------------------------- - - -def test_load_full_config_all_sections(tmp_path): - """load_full_config extracts all sections from mem0.md.""" - from parse_mem0_config import load_full_config - - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text( - """\ -# My Project - -## Retention - -session_state: 90d -decision: forever - -## Search - -default_limit: 20 -boost_recency: true - -## Categories - -- architecture_decisions -- bug_fixes -- security_constraints - -## Identity - -user_id: kartik -project_id: my-project -""", - encoding="utf-8", - ) - - config = load_full_config(str(tmp_path)) - assert config["retention"] == {"session_state": 90, "decision": None} - assert config["search"] == {"default_limit": "20", "boost_recency": "true"} - assert config["categories"] == ["architecture_decisions", "bug_fixes", "security_constraints"] - assert config["identity"] == {"user_id": "kartik", "project_id": "my-project"} - - -def test_load_full_config_partial_sections(tmp_path): - """load_full_config only includes sections that exist.""" - from parse_mem0_config import load_full_config - - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text("## Retention\nsession_state: 30d\n", encoding="utf-8") - - config = load_full_config(str(tmp_path)) - assert "retention" in config - assert "search" not in config - assert "categories" not in config - assert "identity" not in config - - -def test_load_full_config_no_file(tmp_path): - """load_full_config returns {} when no mem0.md exists.""" - from parse_mem0_config import load_full_config - - config = load_full_config(str(tmp_path)) - assert config == {} - - -def test_cli_full_flag(tmp_path): - """CLI: --full prints all sections as JSON.""" - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text( - "## Retention\nsession_state: 90d\n\n## Search\nlimit: 10\n", - encoding="utf-8", - ) - - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "parse_mem0_config.py"), "--full", str(tmp_path)], - capture_output=True, - text=True, - ) - assert result.returncode == 0 - data = json.loads(result.stdout) - assert "retention" in data - assert "search" in data - - -def test_load_full_config_settings_section(tmp_path): - """Settings section is parsed by load_full_config.""" - mem0_md = tmp_path / "mem0.md" - mem0_md.write_text("""\ -## Settings -commit_prompts: true -subagent_skip: Explore, Plan, code-reviewer -""") - sys.path.insert(0, SCRIPTS_DIR) - from parse_mem0_config import load_full_config - config = load_full_config(str(tmp_path)) - assert config.get("settings", {}).get("commit_prompts") == "true" - assert config.get("settings", {}).get("subagent_skip") == "Explore, Plan, code-reviewer" diff --git a/integrations/mem0-plugin/tests/test_project.py b/integrations/mem0-plugin/tests/test_project.py deleted file mode 100644 index 39867b41d..000000000 --- a/integrations/mem0-plugin/tests/test_project.py +++ /dev/null @@ -1,171 +0,0 @@ -"""Tests for _project.py — project_id + branch resolver.""" - -from __future__ import annotations - -import json -import os -import subprocess - - -def test_resolve_project_id_from_https_remote(tmp_git_repo): - from _project import resolve_project_id - - pid = resolve_project_id(str(tmp_git_repo)) - assert pid == "mem0ai-mem0" - - -def test_resolve_project_id_from_ssh_remote(tmp_git_repo_ssh): - from _project import resolve_project_id - - pid = resolve_project_id(str(tmp_git_repo_ssh)) - assert pid == "acme-cool-project" - - -def test_resolve_project_id_fallback_basename(tmp_no_git): - from _project import resolve_project_id - - pid = resolve_project_id(str(tmp_no_git)) - assert pid == os.path.basename(str(tmp_no_git)) - - -def test_resolve_project_id_from_env(tmp_no_git, monkeypatch): - from _project import resolve_project_id - - monkeypatch.setenv("MEM0_PROJECT_ID", "my-override") - pid = resolve_project_id(str(tmp_no_git)) - assert pid == "my-override" - - -def test_resolve_project_id_from_project_map(tmp_no_git): - from _project import resolve_project_id, save_project_mapping - - save_project_mapping(str(tmp_no_git), "custom-project") - pid = resolve_project_id(str(tmp_no_git)) - assert pid == "custom-project" - - -def test_save_project_mapping_creates_file(tmp_no_git): - from _project import save_project_mapping - - save_project_mapping(str(tmp_no_git), "test-proj") - map_path = os.path.expanduser("~/.mem0/project_map.json") - assert os.path.isfile(map_path) - with open(map_path) as f: - data = json.load(f) - assert data[str(tmp_no_git)] == "test-proj" - - -def test_resolve_branch_in_git_repo(tmp_git_repo): - from _project import resolve_branch - - subprocess.run( - ["git", "checkout", "-b", "feat/test-branch"], - cwd=tmp_git_repo, - capture_output=True, - check=True, - ) - branch = resolve_branch(str(tmp_git_repo)) - assert branch == "feat/test-branch" - - -def test_resolve_branch_no_git(tmp_no_git): - from _project import resolve_branch - - branch = resolve_branch(str(tmp_no_git)) - assert branch == "unknown" - - -def test_remote_url_to_slug_various_formats(): - from _project import _remote_url_to_slug - - assert _remote_url_to_slug("https://github.com/mem0ai/mem0.git") == "mem0ai-mem0" - assert _remote_url_to_slug("git@github.com:mem0ai/mem0.git") == "mem0ai-mem0" - assert _remote_url_to_slug("ssh://git@github.com/acme/app.git") == "acme-app" - assert _remote_url_to_slug("https://gitlab.com/org/sub/repo.git") == "sub-repo" - assert _remote_url_to_slug("git@bitbucket.org:team/project.git") == "team-project" - - -def test_remote_url_to_slug_no_git_suffix(): - from _project import _remote_url_to_slug - - assert _remote_url_to_slug("https://github.com/foo/bar") == "foo-bar" - - -def test_resolve_project_id_priority_order(tmp_git_repo, monkeypatch): - """Env var > project_map > git remote > basename.""" - from _project import resolve_project_id, save_project_mapping - - # Git remote gives "mem0ai-mem0" - assert resolve_project_id(str(tmp_git_repo)) == "mem0ai-mem0" - - # project_map overrides git remote - save_project_mapping(str(tmp_git_repo), "from-map") - assert resolve_project_id(str(tmp_git_repo)) == "from-map" - - # Env var overrides everything - monkeypatch.setenv("MEM0_PROJECT_ID", "from-env") - assert resolve_project_id(str(tmp_git_repo)) == "from-env" - - -def test_remote_hash_key_format(tmp_git_repo): - """_remote_hash_key() returns 'remote:<16-char-hex>' for a repo with a remote.""" - import re - - from _project import _remote_hash_key - - key = _remote_hash_key(str(tmp_git_repo)) - assert re.fullmatch(r"remote:[0-9a-f]{16}", key), ( - f"Expected 'remote:<16-char-hex>', got {key!r}" - ) - - -def test_remote_hash_key_no_git(tmp_no_git): - """_remote_hash_key() returns empty string when not in a git repo.""" - from _project import _remote_hash_key - - key = _remote_hash_key(str(tmp_no_git)) - assert key == "" - - -def test_save_project_mapping_writes_remote_key(tmp_git_repo): - """save_project_mapping() writes both the CWD key and the remote hash key.""" - import re - - from _project import save_project_mapping - - save_project_mapping(str(tmp_git_repo), "my-project") - map_path = os.path.expanduser("~/.mem0/project_map.json") - with open(map_path) as f: - data = json.load(f) - - assert data[str(tmp_git_repo)] == "my-project" - remote_keys = [k for k in data if re.fullmatch(r"remote:[0-9a-f]{16}", k)] - assert remote_keys, "Expected at least one remote: key in project_map.json" - assert data[remote_keys[0]] == "my-project" - - -def test_resolve_project_id_remote_hash_fallback(tmp_git_repo, tmp_path): - """Moving the project folder: remote hash key is used as fallback.""" - from _project import resolve_project_id, save_project_mapping - - # Save mapping for original location - save_project_mapping(str(tmp_git_repo), "stable-project") - - # Simulate folder move: resolve using a different CWD path that shares the same remote. - # We use a second tmp_git_repo with the same remote URL to mimic a renamed directory. - import subprocess as _sp - new_repo = tmp_path / "moved_repo" - new_repo.mkdir() - _sp.run(["git", "init"], cwd=new_repo, capture_output=True, check=True) - _sp.run( - ["git", "remote", "add", "origin", "https://github.com/mem0ai/mem0.git"], - cwd=new_repo, - capture_output=True, - check=True, - ) - - # The new CWD is NOT in project_map, but remote hash should match - pid = resolve_project_id(str(new_repo)) - assert pid == "stable-project", ( - f"Expected 'stable-project' via remote hash fallback, got {pid!r}" - ) diff --git a/integrations/mem0-plugin/tests/test_rubric_dedup.py b/integrations/mem0-plugin/tests/test_rubric_dedup.py deleted file mode 100644 index 2a211ddbe..000000000 --- a/integrations/mem0-plugin/tests/test_rubric_dedup.py +++ /dev/null @@ -1,61 +0,0 @@ -"""Tests for rubric deduplication in on_user_prompt.sh.""" - -from __future__ import annotations - -import json -import os -import subprocess - -import pytest - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - - -@pytest.fixture(autouse=True) -def _clean_rubric_flag(tmp_path, monkeypatch): - """Use a temp dir for the rubric flag file and clean msg counter.""" - monkeypatch.setenv("MEM0_RUBRIC_DIR", str(tmp_path)) - msg_count_file = "/tmp/mem0_msg_count_testuser" - yield - if os.path.exists(msg_count_file): - os.unlink(msg_count_file) - - -def _run_hook(prompt: str, env_overrides: dict | None = None, session_id: str = "test-sess-001") -> str: - """Run on_user_prompt.sh with a simulated prompt and return stdout.""" - env = { - **os.environ, - "USER": "testuser", - "MEM0_API_KEY": "test-key-123", - "MEM0_RESOLVED_USER_ID": "testuser", - "MEM0_PROJECT_ID": "test-project", - "MEM0_BRANCH": "main", - "MEM0_PREFETCH": "false", - } - if env_overrides: - env.update(env_overrides) - - input_json = json.dumps({"prompt": prompt, "session_id": session_id}) - result = subprocess.run( - ["bash", os.path.join(SCRIPTS_DIR, "on_user_prompt.sh")], - input=input_json, - capture_output=True, - text=True, - env=env, - timeout=10, - ) - return result.stdout - - -def test_first_prompt_gets_full_rubric(): - """First substantial prompt of session gets full memory check rubric.""" - output = _run_hook("How should we refactor the auth module?") - assert "Mem0 searches apply" in output - assert "metadata.type" in output - - -def test_second_prompt_gets_no_rubric(): - """Second prompt of session emits nothing — rubric and tips only on first prompt.""" - _run_hook("How should we refactor the auth module?") - output = _run_hook("What about the database layer?") - assert output.strip() == "" diff --git a/integrations/mem0-plugin/tests/test_search.py b/integrations/mem0-plugin/tests/test_search.py deleted file mode 100644 index 58c1c8042..000000000 --- a/integrations/mem0-plugin/tests/test_search.py +++ /dev/null @@ -1,211 +0,0 @@ -"""Tests for _search.py — shared mem0 search API helper.""" - -from __future__ import annotations - -import json -import urllib.error -from unittest.mock import MagicMock, patch - - -def test_search_memories_returns_results(): - from _search import search_memories - - fake_results = [ - {"id": "abc123", "memory": "Use Postgres for auth", "metadata": {"type": "decision"}}, - {"id": "def456", "memory": "Never use floats for money", "metadata": {"type": "anti_pattern"}}, - ] - - def mock_urlopen(req, timeout=None): - resp = MagicMock() - resp.read.return_value = json.dumps({"results": fake_results}).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - results = search_memories("test-key", "user1", "proj1", "auth decisions") - - assert len(results) == 2 - assert results[0]["id"] == "abc123" - - -def test_search_memories_with_metadata_type(): - from _search import search_memories - - captured_body = {} - - def mock_urlopen(req, timeout=None): - captured_body.update(json.loads(req.data.decode())) - resp = MagicMock() - resp.read.return_value = json.dumps({"results": []}).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - search_memories("key", "user", "proj", "query", metadata_type="decision") - - filters = captured_body["filters"] - assert {"metadata": {"type": "decision"}} in filters["AND"] - - -def test_search_memories_handles_api_error(): - from _search import search_memories - - with patch("urllib.request.urlopen", side_effect=Exception("timeout")): - results = search_memories("key", "user", "proj", "query") - - assert results == [] - - -def test_search_memories_handles_list_response(): - from _search import search_memories - - fake_results = [{"id": "abc", "memory": "test"}] - - def mock_urlopen(req, timeout=None): - resp = MagicMock() - resp.read.return_value = json.dumps(fake_results).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - results = search_memories("key", "user", "proj", "query") - - assert len(results) == 1 - - -def test_search_memories_respects_top_k(): - from _search import search_memories - - captured_body = {} - - def mock_urlopen(req, timeout=None): - captured_body.update(json.loads(req.data.decode())) - resp = MagicMock() - resp.read.return_value = json.dumps({"results": []}).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - search_memories("key", "user", "proj", "query", top_k=5) - - assert captured_body["top_k"] == 5 - - -def test_search_memories_no_api_key_returns_empty(): - from _search import search_memories - - results = search_memories("", "user", "proj", "query") - assert results == [] - - -def test_search_memories_logs_rate_limit_error(capsys): - """Bug bash #22: a 429 must not look identical to a genuine empty result.""" - from _search import search_memories - - def mock_urlopen(req, timeout=None): - raise urllib.error.HTTPError("http://x", 429, "Too Many Requests", {}, None) - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - results = search_memories("key", "user", "proj", "query") - - assert results == [] - err = capsys.readouterr().err - assert "429" in err - assert "Too Many Requests" in err - - -def test_search_memories_happy_path_is_silent(capsys): - from _search import search_memories - - def mock_urlopen(req, timeout=None): - resp = MagicMock() - resp.read.return_value = json.dumps({"results": []}).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - results = search_memories("key", "user", "proj", "query") - - assert results == [] - assert capsys.readouterr().err == "" - - -def test_search_memories_omits_rerank_by_default(): - """Regression for #5684: rerank must not be sent unless requested.""" - from _search import search_memories - - captured_body = {} - - def mock_urlopen(req, timeout=None): - captured_body.update(json.loads(req.data.decode())) - resp = MagicMock() - resp.read.return_value = json.dumps({"results": []}).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - search_memories("key", "user", "proj", "query") - - assert "rerank" not in captured_body - - -def test_search_memories_forwards_rerank_true(): - """Regression for #5684: rerank=True must reach the request body so the - REST endpoint actually reranks (it does not rerank when omitted).""" - from _search import search_memories - - captured_body = {} - - def mock_urlopen(req, timeout=None): - captured_body.update(json.loads(req.data.decode())) - resp = MagicMock() - resp.read.return_value = json.dumps({"results": []}).encode() - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - search_memories("key", "user", "proj", "query", rerank=True) - - assert captured_body.get("rerank") is True - - -def test_should_rerank_defaults_true(monkeypatch): - """Regression for #5684: auto-injection reranks by default.""" - from _search import should_rerank - - monkeypatch.delenv("MEM0_RERANK", raising=False) - assert should_rerank() is True - - -def test_should_rerank_opt_out_values(monkeypatch): - from _search import should_rerank - - for falsey in ("0", "false", "False", "NO", "off", ""): - monkeypatch.setenv("MEM0_RERANK", falsey) - assert should_rerank() is False, falsey - - for truthy in ("1", "true", "yes", "on"): - monkeypatch.setenv("MEM0_RERANK", truthy) - assert should_rerank() is True, truthy - - -def test_format_results_for_context(): - from _search import format_results_for_context - - memories = [ - {"id": "abc12345-long-id", "memory": "Use Postgres for auth", "metadata": {"type": "decision"}}, - {"id": "def67890-long-id", "memory": "JWT tokens expire in 1h", "metadata": {"type": "convention"}}, - ] - - output = format_results_for_context(memories, heading="Relevant memories") - assert "Relevant memories" in output - assert "[decision]" in output - assert "Use Postgres for auth" in output - assert "abc12345" in output diff --git a/integrations/mem0-plugin/tests/test_session_stats.py b/integrations/mem0-plugin/tests/test_session_stats.py deleted file mode 100644 index 19d15e20d..000000000 --- a/integrations/mem0-plugin/tests/test_session_stats.py +++ /dev/null @@ -1,248 +0,0 @@ -"""Tests for session_stats.py — session-level memory operation tracker.""" - -from __future__ import annotations - -import json -import os -import subprocess -import sys - -import pytest - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") - - -@pytest.fixture(autouse=True) -def _isolate_stats_file(tmp_path, monkeypatch): - """Point STATS_FILE to a temp location so tests don't interfere.""" - stats_file = str(tmp_path / "test_stats.json") - monkeypatch.setattr("session_stats.STATS_FILE", stats_file) - yield stats_file - - -def test_init_creates_file(_isolate_stats_file): - import session_stats - - session_stats.init() - assert os.path.isfile(_isolate_stats_file) - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["adds"] == 0 - assert data["searches"] == 0 - assert data["categories"] == [] - - -def test_record_add_increments(_isolate_stats_file): - import session_stats - - session_stats.init() - session_stats.record_add("bug_fixes") - session_stats.record_add("bug_fixes") - session_stats.record_add("decisions") - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["adds"] == 3 - assert set(data["categories"]) == {"bug_fixes", "decisions"} - - -def test_record_search_increments(_isolate_stats_file): - import session_stats - - session_stats.init() - session_stats.record_search() - session_stats.record_search() - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["searches"] == 2 - - -def test_report_returns_summary(_isolate_stats_file): - import session_stats - - session_stats.init() - session_stats.record_add("architecture_decisions") - session_stats.record_add("task_learnings") - session_stats.record_search() - session_stats.record_search() - session_stats.record_search() - - result = session_stats.report() - assert "wrote 2 memories" in result - assert "retrieved 3" in result - assert "architecture_decisions" in result - assert "task_learnings" in result - - -def test_report_empty_session(_isolate_stats_file): - import session_stats - - session_stats.init() - result = session_stats.report() - assert result == "" - - -def test_report_preserves_file(_isolate_stats_file): - import session_stats - - session_stats.init() - session_stats.record_add() - session_stats.report() - assert os.path.isfile(_isolate_stats_file) - - -def test_record_add_no_category(_isolate_stats_file): - import session_stats - - session_stats.init() - session_stats.record_add("") - session_stats.record_add() - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["adds"] == 2 - assert data["categories"] == [] - - -def test_duplicate_categories_not_added(_isolate_stats_file): - import session_stats - - session_stats.init() - session_stats.record_add("bug_fixes") - session_stats.record_add("bug_fixes") - session_stats.record_add("bug_fixes") - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["adds"] == 3 - assert data["categories"] == ["bug_fixes"] - - -def test_cli_init(tmp_path): - """Test CLI invocation: session_stats.py init.""" - env = {**os.environ, "USER": "test"} - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "session_stats.py"), "init"], - capture_output=True, - text=True, - env=env, - ) - assert result.returncode == 0 - - -def test_cli_report_no_data(tmp_path): - """Test CLI invocation: report with no prior init prints fallback.""" - env = {**os.environ, "USER": f"test_{os.getpid()}"} - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "session_stats.py"), "report"], - capture_output=True, - text=True, - env=env, - ) - assert result.returncode == 0 - - -def test_peek_returns_json_without_clearing(_isolate_stats_file): - """peek returns JSON stats without deleting the stats file.""" - import session_stats - - session_stats.init() - session_stats.record_add("decisions") - session_stats.record_add("decisions") - session_stats.record_search() - - result = session_stats.peek() - data = json.loads(result) - assert data["adds"] == 2 - assert data["searches"] == 1 - - assert os.path.isfile(_isolate_stats_file) - - -def test_category_counts_tracked(_isolate_stats_file): - """category_counts tracks per-category add counts.""" - import session_stats - - session_stats.init() - session_stats.record_add("bug_fixes") - session_stats.record_add("bug_fixes") - session_stats.record_add("bug_fixes") - session_stats.record_add("decisions") - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["category_counts"]["bug_fixes"] == 3 - assert data["category_counts"]["decisions"] == 1 - - -def test_category_counts_empty_category_not_tracked(_isolate_stats_file): - """Empty category string doesn't appear in category_counts.""" - import session_stats - - session_stats.init() - session_stats.record_add("") - session_stats.record_add() - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["category_counts"] == {} - - -def test_recent_ids_tracked(_isolate_stats_file): - """record_add with memory_id stores ID in recent_ids.""" - import session_stats - - session_stats.init() - session_stats.record_add("decision", "abc-123") - session_stats.record_add("convention", "def-456") - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert len(data["recent_ids"]) == 2 - assert data["recent_ids"][0]["id"] == "abc-123" - assert data["recent_ids"][1]["id"] == "def-456" - assert data["recent_ids"][0]["category"] == "decision" - - -def test_recent_ids_capped(_isolate_stats_file): - """recent_ids list is capped at MAX_RECENT_IDS.""" - import session_stats - - session_stats.init() - for i in range(60): - session_stats.record_add("test", f"id-{i}") - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert len(data["recent_ids"]) == session_stats.MAX_RECENT_IDS - assert data["recent_ids"][0]["id"] == f"id-{60 - session_stats.MAX_RECENT_IDS}" - - -def test_recent_ids_empty_without_memory_id(_isolate_stats_file): - """record_add without memory_id doesn't add to recent_ids.""" - import session_stats - - session_stats.init() - session_stats.record_add("decision") - session_stats.record_add("convention", "") - - with open(_isolate_stats_file) as f: - data = json.load(f) - assert data["recent_ids"] == [] - - -def test_cli_peek(tmp_path): - """Test CLI invocation: session_stats.py peek outputs JSON.""" - env = {**os.environ, "USER": "test"} - subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "session_stats.py"), "init"], - capture_output=True, text=True, env=env, - ) - result = subprocess.run( - [sys.executable, os.path.join(SCRIPTS_DIR, "session_stats.py"), "peek"], - capture_output=True, text=True, env=env, - ) - assert result.returncode == 0 - data = json.loads(result.stdout) - assert "adds" in data diff --git a/integrations/mem0-plugin/tests/test_telemetry.py b/integrations/mem0-plugin/tests/test_telemetry.py deleted file mode 100644 index c44cc1b9b..000000000 --- a/integrations/mem0-plugin/tests/test_telemetry.py +++ /dev/null @@ -1,268 +0,0 @@ -"""Tests for telemetry.py — fire-and-forget PostHog plugin telemetry.""" - -from __future__ import annotations - -import json -import os -import sys -import urllib.error - -SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") -sys.path.insert(0, os.path.abspath(SCRIPTS_DIR)) - - -def test_import_succeeds(): - import telemetry - - assert hasattr(telemetry, "emit") - assert hasattr(telemetry, "main") - - -def test_opt_out_skips_send(monkeypatch): - import telemetry - - monkeypatch.setenv("MEM0_TELEMETRY", "false") - sent = [] - monkeypatch.setattr(telemetry, "send", lambda p: sent.append(p)) - telemetry.emit("session_start") - assert sent == [] - - -def test_opt_out_variants(monkeypatch): - import telemetry - - for val in ("0", "no", "off", "FALSE", "No"): - monkeypatch.setenv("MEM0_TELEMETRY", val) - assert not telemetry.is_enabled() - - -def test_enabled_by_default(monkeypatch): - import telemetry - - monkeypatch.delenv("MEM0_TELEMETRY", raising=False) - assert telemetry.is_enabled() - - -def test_posthog_payload_structure(monkeypatch): - import telemetry - - monkeypatch.setenv("MEM0_RESOLVED_USER_ID", "testuser") - monkeypatch.setenv("MEM0_PROJECT_ID", "test-project") - monkeypatch.delenv("MEM0_API_KEY", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", raising=False) - - payload = telemetry.build_posthog_payload("plugin.session_start", {"memory_count": 5}) - - assert payload["api_key"] == telemetry.POSTHOG_API_KEY - assert payload["event"] == "plugin.session_start" - assert "distinct_id" in payload - assert payload["properties"]["source"] == "plugin" - assert isinstance(payload["properties"]["plugin_version"], str) - assert payload["properties"]["plugin_version"] != "" - assert payload["properties"]["memory_count"] == 5 - assert payload["properties"]["$process_person_profile"] is False - - raw = json.dumps(payload) - assert "testuser" not in raw - assert "test-project" not in raw - - -def test_system_props_override_caller_props(monkeypatch): - """H8: system properties must win over caller-supplied properties.""" - import telemetry - - monkeypatch.setenv("MEM0_RESOLVED_USER_ID", "testuser") - monkeypatch.setenv("MEM0_PROJECT_ID", "test-project") - monkeypatch.delenv("MEM0_API_KEY", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", raising=False) - - # Caller tries to override system-controlled properties - caller_props = { - "source": "CALLER_OVERRIDE", - "platform": "CALLER_OVERRIDE", - "plugin_version": "CALLER_OVERRIDE", - "memory_count": 42, - } - payload = telemetry.build_posthog_payload("plugin.test", caller_props) - props = payload["properties"] - - # System props must win - assert props["source"] == "plugin" - assert props["platform"] == telemetry.detect_platform() - assert props["plugin_version"] == telemetry._load_plugin_version(telemetry.detect_platform()) - # Caller-only props still present - assert props["memory_count"] == 42 - - -def test_distinct_id_from_api_key(monkeypatch): - import hashlib - - import telemetry - - monkeypatch.setenv("MEM0_API_KEY", "m0-testkey123") - expected = hashlib.sha256(b"m0-testkey123").hexdigest()[:32] - assert telemetry._distinct_id() == expected - - -def test_distinct_id_fallback_no_key(monkeypatch): - import telemetry - - monkeypatch.delenv("MEM0_API_KEY", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", raising=False) - monkeypatch.setenv("MEM0_RESOLVED_USER_ID", "kartik") - assert telemetry._distinct_id() == telemetry._sha256("kartik") - - -def test_hash_deterministic(): - import telemetry - - h1 = telemetry._sha256("same-value") - h2 = telemetry._sha256("same-value") - assert h1 == h2 - assert h1 != telemetry._sha256("different-value") - - -def test_platform_cursor(monkeypatch): - import telemetry - - monkeypatch.delenv("MEM0_PLATFORM", raising=False) - monkeypatch.delenv("ANTIGRAVITY_PLUGIN_ROOT", raising=False) - monkeypatch.delenv("PLUGIN_ROOT", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_ROOT", raising=False) - monkeypatch.setenv("CURSOR_PLUGIN_ROOT", "/path") - assert telemetry.detect_platform() == "cursor" - - -def test_platform_codex(monkeypatch): - import telemetry - - monkeypatch.delenv("MEM0_PLATFORM", raising=False) - monkeypatch.delenv("ANTIGRAVITY_PLUGIN_ROOT", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_ROOT", raising=False) - monkeypatch.delenv("CURSOR_PLUGIN_ROOT", raising=False) - monkeypatch.setenv("PLUGIN_ROOT", "/path") - assert telemetry.detect_platform() == "codex" - - -def test_platform_kimi(monkeypatch): - """Kimi Code usage must be attributed to its own platform so it can be - counted distinctly. The shim also pins MEM0_PLATFORM=kimi; this covers the - KIMI_PLUGIN_ROOT auto-detection fallback.""" - import telemetry - - monkeypatch.delenv("MEM0_PLATFORM", raising=False) - monkeypatch.delenv("ANTIGRAVITY_PLUGIN_ROOT", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_ROOT", raising=False) - monkeypatch.delenv("CURSOR_PLUGIN_ROOT", raising=False) - monkeypatch.delenv("PLUGIN_ROOT", raising=False) - monkeypatch.setenv("KIMI_PLUGIN_ROOT", "/path") - assert telemetry.detect_platform() == "kimi" - - -def test_platform_explicit_override(monkeypatch): - """MEM0_PLATFORM wins over auto-detection so each editor can label - itself reliably even when host env vars are ambiguous or absent.""" - import telemetry - - monkeypatch.setenv("CLAUDE_PLUGIN_ROOT", "/path") # conflicting auto-signal - monkeypatch.setenv("MEM0_PLATFORM", "cursor") - assert telemetry.detect_platform() == "cursor" - - -def test_platform_antigravity(monkeypatch): - """Antigravity sets CLAUDE_PLUGIN_ROOT so the shared scripts resolve their - paths. That must not change how it is attributed.""" - import telemetry - - monkeypatch.delenv("MEM0_PLATFORM", raising=False) - monkeypatch.setenv("ANTIGRAVITY_PLUGIN_ROOT", "/ext") - monkeypatch.setenv("CLAUDE_PLUGIN_ROOT", "/ext") # antigravity sets both - assert telemetry.detect_platform() == "antigravity" - - -def test_plugin_version_is_per_editor(monkeypatch): - """Each editor reports the version from its OWN manifest. Antigravity is on - a 0.1.x line while Cursor/Codex are on 0.2.x, so they must not all report - the same shared version. An unsupported surface reports "unknown" rather - than borrowing a version it does not ship.""" - import telemetry - - plugin_dir = os.path.join(os.path.dirname(__file__), "..") - manifests = { - "antigravity": "plugin.json", - "cursor": os.path.join(".cursor-plugin", "plugin.json"), - "codex": os.path.join(".codex-plugin", "plugin.json"), - "kimi": os.path.join(".kimi-plugin", "plugin.json"), - } - for plat, rel in manifests.items(): - monkeypatch.setenv("MEM0_PLATFORM", plat) - with open(os.path.join(plugin_dir, rel)) as f: - expected = json.load(f)["version"] - payload = telemetry.build_posthog_payload("plugin.test") - assert payload["properties"]["plugin_version"] == expected, f"{plat} should report {expected} from {rel}" - - monkeypatch.setenv("MEM0_PLATFORM", "claude-code") - payload = telemetry.build_posthog_payload("plugin.test") - assert payload["properties"]["plugin_version"] == "unknown" - - -def test_send_fails_silently(monkeypatch): - import telemetry - - def raise_error(req, timeout): - raise urllib.error.URLError("connection refused") - - monkeypatch.setattr(telemetry.urllib.request, "urlopen", raise_error) - telemetry.send({"event": "test"}) - - -def test_cli_exits_zero_when_disabled(monkeypatch): - import telemetry - - monkeypatch.setenv("MEM0_TELEMETRY", "false") - monkeypatch.setattr(sys, "argv", ["telemetry.py", "session_start"]) - assert telemetry.main() == 0 - - -def test_cli_no_args_exits_nonzero(monkeypatch): - import telemetry - - monkeypatch.delenv("MEM0_TELEMETRY", raising=False) - monkeypatch.setattr(sys, "argv", ["telemetry.py"]) - assert telemetry.main() == 1 - - -def test_cursor_wrappers_pin_platform(): - """Cursor wrappers delegate to the shared scripts, which auto-detect the - platform from host env vars. Since Cursor may not export CURSOR_PLUGIN_ROOT - to the subprocess, each wrapper must pin MEM0_PLATFORM=cursor so the - delegated telemetry is attributed to cursor, not the 'plugin' fallback.""" - scripts_dir = os.path.join(os.path.dirname(__file__), "..", "scripts") - cursor_wrappers = [ - "on_session_start_cursor.sh", - "on_user_prompt_cursor.sh", - "on_post_tool_use_cursor.sh", - "on_pre_compact_cursor.sh", - "on_stop_cursor.sh", - ] - for name in cursor_wrappers: - with open(os.path.join(scripts_dir, name)) as f: - content = f.read() - assert "export MEM0_PLATFORM=cursor" in content, f"{name} must `export MEM0_PLATFORM=cursor` before delegating" - - -def test_codex_hooks_pin_platform(): - """Codex installs standalone hooks (absolute paths) via install_codex_hooks.py, - so PLUGIN_ROOT is not set at runtime and the platform falls back to 'plugin'. - Codex runs hook commands through a shell, so every command pins - MEM0_PLATFORM=codex inline for correct attribution.""" - hooks_path = os.path.join(os.path.dirname(__file__), "..", "hooks", "codex-hooks.json") - with open(hooks_path) as f: - config = json.load(f) - - commands = [ - h["command"] for entries in config["hooks"].values() for entry in entries for h in entry.get("hooks", []) - ] - assert commands, "expected at least one codex hook command" - for cmd in commands: - assert "MEM0_PLATFORM=codex" in cmd, f"codex hook command missing platform pin: {cmd}" diff --git a/integrations/mem0-plugin/tests/test_write_path.py b/integrations/mem0-plugin/tests/test_write_path.py deleted file mode 100644 index 1be57df83..000000000 --- a/integrations/mem0-plugin/tests/test_write_path.py +++ /dev/null @@ -1,207 +0,0 @@ -"""Tests for write-path app_id migration and API key resolution. - -Verifies that all scripts writing to the Mem0 API: -1. Pass app_id as a top-level parameter (not in metadata) -2. Do NOT include project_id in metadata -3. Include branch in metadata when available -4. Use resolve_api_key() for key resolution with userConfig fallback -""" - -from __future__ import annotations - -import json -from unittest.mock import MagicMock, patch - - -def test_auto_import_post_memory_uses_app_id(): - """auto_import.post_memory sends app_id top-level, not metadata.project_id.""" - from auto_import import post_memory - - captured = {} - - def mock_urlopen(req, timeout=None): - body = json.loads(req.data.decode("utf-8")) - captured.update(body) - resp = MagicMock() - resp.status = 200 - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - result = post_memory( - api_key="test-key", - content="test content", - user_id="testuser", - filename="CLAUDE.md", - project_id="my-project", - branch="main", - ) - - assert result is True - assert captured["app_id"] == "my-project" - assert captured["user_id"] == "testuser" - assert "project_id" not in captured.get("metadata", {}) - assert captured["metadata"]["type"] == "project_profile" - assert captured["metadata"]["branch"] == "main" - assert captured["infer"] is False - - -def test_auto_import_post_memory_omits_empty_branch(): - """auto_import.post_memory skips branch in metadata when empty.""" - from auto_import import post_memory - - captured = {} - - def mock_urlopen(req, timeout=None): - body = json.loads(req.data.decode("utf-8")) - captured.update(body) - resp = MagicMock() - resp.status = 200 - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - post_memory("key", "content", "user", "FILE.md", "proj", branch="") - - assert "branch" not in captured.get("metadata", {}) - - -def test_on_pre_compact_store_memory_uses_app_id(): - """on_pre_compact.store_memory sends app_id top-level.""" - from on_pre_compact import store_memory - - captured = {} - - def mock_urlopen(req, timeout=None): - body = json.loads(req.data.decode("utf-8")) - captured.update(body) - resp = MagicMock() - resp.status = 200 - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - result = store_memory( - api_key="test-key", - content="session state content", - user_id="testuser", - source="pre-compaction", - session_id="sess-123", - project_id="my-project", - branch="feat/auth", - ) - - assert result is True - assert captured["app_id"] == "my-project" - assert captured["user_id"] == "testuser" - assert "project_id" not in captured.get("metadata", {}) - assert captured["metadata"]["type"] == "session_state" - assert captured["metadata"]["source"] == "pre-compaction" - assert captured["metadata"]["branch"] == "feat/auth" - assert "expiration_date" in captured - - -def test_capture_compact_summary_store_uses_app_id(): - """capture_compact_summary.store_summary sends app_id top-level.""" - from capture_compact_summary import store_summary - - captured = {} - - def mock_urlopen(req, timeout=None): - body = json.loads(req.data.decode("utf-8")) - captured.update(body) - resp = MagicMock() - resp.status = 200 - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - result = store_summary( - api_key="test-key", - summary="compact summary text", - user_id="testuser", - session_id="sess-456", - project_id="my-project", - branch="main", - ) - - assert result is True - assert captured["app_id"] == "my-project" - assert captured["user_id"] == "testuser" - assert "project_id" not in captured.get("metadata", {}) - assert captured["metadata"]["type"] == "compact_summary" - assert captured["metadata"]["branch"] == "main" - assert captured["infer"] is True - assert "expiration_date" in captured - - -def test_no_metadata_project_id_anywhere(): - """Ensure none of the write functions put project_id in metadata.""" - from auto_import import post_memory - from capture_compact_summary import store_summary - from on_pre_compact import store_memory - - bodies = [] - - def mock_urlopen(req, timeout=None): - body = json.loads(req.data.decode("utf-8")) - bodies.append(body) - resp = MagicMock() - resp.status = 200 - resp.__enter__ = lambda s: s - resp.__exit__ = MagicMock(return_value=False) - return resp - - with patch("urllib.request.urlopen", side_effect=mock_urlopen): - post_memory("k", "c", "u", "f", "proj", "br") - store_memory("k", "c", "u", "src", "sid", "proj", "br") - store_summary("k", "s", "u", "sid", "proj", "br") - - for i, body in enumerate(bodies): - metadata = body.get("metadata", {}) - assert "project_id" not in metadata, f"Write function #{i} still has metadata.project_id" - assert body.get("app_id") == "proj", f"Write function #{i} missing app_id top-level" - - -def test_resolve_api_key_prefers_env_var(monkeypatch): - """resolve_api_key returns MEM0_API_KEY when both are set.""" - from _identity import resolve_api_key - - monkeypatch.setenv("MEM0_API_KEY", "direct-key") - monkeypatch.setenv("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", "fallback-key") - assert resolve_api_key() == "direct-key" - - -def test_resolve_api_key_falls_back_to_plugin_option(monkeypatch): - """resolve_api_key falls back to CLAUDE_PLUGIN_OPTION_MEM0_API_KEY.""" - from _identity import resolve_api_key - - monkeypatch.delenv("MEM0_API_KEY", raising=False) - monkeypatch.setenv("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", "fallback-key") - assert resolve_api_key() == "fallback-key" - - -def test_resolve_api_key_returns_empty_when_neither_set(monkeypatch): - """resolve_api_key returns empty string when no key is available.""" - from _identity import resolve_api_key - - monkeypatch.delenv("MEM0_API_KEY", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_OPTION_API_KEY", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", raising=False) - monkeypatch.setattr("_identity._extract_key_from_shell_profiles", lambda: "") - assert resolve_api_key() == "" - - -def test_resolve_api_key_falls_back_to_shell_profile(monkeypatch): - """resolve_api_key extracts key from shell profile when env vars are empty.""" - from _identity import resolve_api_key - - monkeypatch.delenv("MEM0_API_KEY", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_OPTION_API_KEY", raising=False) - monkeypatch.delenv("CLAUDE_PLUGIN_OPTION_MEM0_API_KEY", raising=False) - monkeypatch.setattr("_identity._extract_key_from_shell_profiles", lambda: "m0-from-profile") - assert resolve_api_key() == "m0-from-profile" diff --git a/integrations/mem0-strands/python/pyproject.toml b/integrations/mem0-strands/python/pyproject.toml index 5b558993f..5776ad695 100644 --- a/integrations/mem0-strands/python/pyproject.toml +++ b/integrations/mem0-strands/python/pyproject.toml @@ -77,7 +77,7 @@ select = [ ] [tool.mypy] -python_version = "3.10" +python_version = "3.12" warn_return_any = true warn_unused_configs = true ignore_missing_imports = true diff --git a/integrations/openclaw/README.md b/integrations/openclaw/README.md index 4db861feb..1d70dffa3 100644 --- a/integrations/openclaw/README.md +++ b/integrations/openclaw/README.md @@ -4,7 +4,7 @@ Long-term memory for [OpenClaw](https://github.com/openclaw/openclaw) agents, po Your agent forgets everything between sessions. This plugin fixes that — it stores conversations, extracts what matters, and brings it back when relevant. -By default, the plugin runs in **skills mode**: the agent controls what to remember (triage), how to recall (recall), and periodic cleanup (dream). Skills mode, `autoRecall`, and `autoCapture` are all enabled by default during `openclaw mem0 init`. +By default, the plugin runs in **skills mode**: the agent controls what to remember (triage) and how to recall (recall). Skills mode, `autoRecall`, and `autoCapture` are all enabled by default during `openclaw mem0 init`. ## Requirements @@ -74,7 +74,6 @@ Humans should follow the Quick Start below. "keywordSearch": true, "identityAlwaysInclude": true }, - "dream": { "enabled": true }, "domain": "companion" } } @@ -210,11 +209,10 @@ All `oss` fields are optional. See the [Mem0 OSS docs](https://docs.mem0.ai/open ### Skills Mode (Default) -Enabled automatically during `openclaw mem0 init`. The agent controls memory through three skills: +Enabled automatically during `openclaw mem0 init`. The agent controls memory through two skills: - **Triage** — Extracts durable facts from conversations using a structured protocol. Categories, importance gates, and domain overlays control what gets stored. - **Recall** — Before each turn, rewrites the user message into search queries, retrieves relevant memories with reranking, and injects them into context. -- **Dream** — Periodic memory consolidation: merges duplicates, resolves conflicts, and prunes stale entries. When skills mode is active, the skills handle memory operations. `autoRecall` and `autoCapture` remain `true` by default alongside skills mode. The built-in `session-memory` hook is disabled to avoid conflicts. @@ -280,10 +278,6 @@ openclaw mem0 config set user_id alice openclaw mem0 event list openclaw mem0 event status -# Memory consolidation -openclaw mem0 dream -openclaw mem0 dream --dry-run - # JSON output (any command) openclaw mem0 search "preferences" --json openclaw mem0 list --json @@ -316,7 +310,6 @@ Enabled by default during `openclaw mem0 init`. `autoRecall` and `autoCapture` a | `skills.recall.rerank` | `boolean` | `true` | Rerank search results for relevance | | `skills.recall.keywordSearch` | `boolean` | `true` | Augment with keyword-based search | | `skills.recall.identityAlwaysInclude` | `boolean` | `true` | Always include identity memories | -| `skills.dream.enabled` | `boolean` | `true` | Enable periodic memory consolidation | | `skills.domain` | `string` | `"companion"` | Domain overlay for triage rules | ### Platform Mode @@ -362,7 +355,7 @@ To avoid plaintext credentials: ### Memory Processing -In **skills mode** (default after `openclaw mem0 init`), the agent uses structured protocols (triage, recall, dream) to decide what to store and recall. The built-in `session-memory` hook is disabled to avoid conflicts. +In **skills mode** (default after `openclaw mem0 init`), the agent uses structured triage and recall protocols to decide what to store and recall. The built-in `session-memory` hook is disabled to avoid conflicts. Without skills, `autoCapture` and `autoRecall` are both enabled by default: - `autoCapture`: sends conversation content to your configured backend after each agent turn @@ -377,7 +370,6 @@ In platform mode, conversation content is sent to `api.mem0.ai` for processing. | `~/.openclaw/openclaw.json` | Plugin configuration (API keys, user ID, settings) | | `~/.mem0/vector_store.db` | Local vector store (open-source mode only) | | `~/.mem0/history.db` | Memory edit history (open-source mode only) | -| `/dream-state.json` | Memory consolidation state | ## License diff --git a/integrations/openclaw/cli/commands.ts b/integrations/openclaw/cli/commands.ts index 2d5897446..e2eb71bf5 100644 --- a/integrations/openclaw/cli/commands.ts +++ b/integrations/openclaw/cli/commands.ts @@ -20,7 +20,6 @@ * - config set : Update a plugin config field * - event list : List recent background events * - event status: Get status of a specific event - * - dream : Run memory consolidation * * Naming conventions match the Python CLI (`mem0 init`, `mem0 search`, etc.) */ @@ -36,7 +35,6 @@ import type { MemoryItem, SearchOptions, } from "../types.ts"; -import { loadDreamPrompt } from "../skill-loader.ts"; import { readText } from "../fs-safe.ts"; import type { PluginAuthConfig } from "./config-file.ts"; import { @@ -1227,7 +1225,18 @@ export function registerCliCommands( .option("--json", "Output as JSON") .action(async (opts: { json?: boolean } = {}) => { try { - const auth = readPluginAuth(); + if (cfg.needsSetup || !backend) { + const error = "Mem0 is not configured. Run `openclaw mem0 init`."; + if (jsonOut(opts, { + ok: true, + mode: cfg.mode, + connected: false, + userId: cfg.userId, + error, + })) return; + console.log(error); + return; + } const result = await backend.status(); if (jsonOut(opts, { ok: true, @@ -1714,7 +1723,6 @@ export function registerCliCommands( status: "Check connectivity and authentication", config: "Manage mem0 configuration (show, get, set)", event: "Manage background processing events (list, status)", - dream: "Run memory consolidation (review, merge, prune)", help: "Show help. Use --json for machine-readable output (for LLM agents)", }, }; @@ -1736,7 +1744,6 @@ export function registerCliCommands( status: { description: "Check connectivity", flags: { "--json": "JSON output" } }, config: { description: "Manage configuration (show, get, set)", flags: { "--json": "JSON output" } }, event: { description: "Manage background events (list, status)", flags: { "--json": "JSON output" } }, - dream: { description: "Run memory consolidation", flags: { "--dry-run": "Show inventory only", "--json": "JSON output" } }, help: { description: "Show help", flags: { "--json": "JSON output" } }, }, }, @@ -1760,108 +1767,6 @@ export function registerCliCommands( console.log(""); }); - // ==================================================================== - // dream - // ==================================================================== - - mem0 - .command("dream") - .description( - "Run memory consolidation (review, merge, prune stored memories)", - ) - .option( - "--dry-run", - "Show memory inventory without running consolidation", - ) - .option("--json", "Output as JSON") - .action(async (opts: { dryRun?: boolean; json?: boolean }) => { - try { - const uid = cfg.userId; - const memories = await provider.getAll({ - user_id: uid, - source: "OPENCLAW", - }); - const count = Array.isArray(memories) ? memories.length : 0; - - if (count === 0) { - if (jsonOut(opts, { ok: true, count: 0, message: "No memories to consolidate." })) return; - console.log("No memories to consolidate."); - return; - } - - const catCounts = new Map(); - for (const mem of memories) { - const cat = - (mem.metadata as any)?.category ?? - mem.categories?.[0] ?? - "uncategorized"; - catCounts.set(cat, (catCounts.get(cat) ?? 0) + 1); - } - - if (opts.dryRun && opts.json) { - jsonOut(opts, { ok: true, count, categories: Object.fromEntries(catCounts) }); - return; - } - - if (opts.json && !opts.dryRun) { - jsonOut(opts, { ok: true, count, message: `${count} memories available for consolidation` }); - return; - } - - process.stderr.write(`\nMemory inventory for "${uid}":\n`); - for (const [cat, num] of [...catCounts.entries()].sort( - (a, b) => b[1] - a[1], - )) { - process.stderr.write(` ${cat}: ${num}\n`); - } - process.stderr.write(` TOTAL: ${count}\n\n`); - - if (opts.dryRun) { - process.stderr.write("Dry run — no changes made.\n"); - return; - } - - const dreamPrompt = loadDreamPrompt(cfg.skills ?? {}); - if (!dreamPrompt) { - process.stderr.write( - "Dream skill file not found at skills/memory-dream/SKILL.md\n", - ); - return; - } - - const memoryDump = (memories as MemoryItem[]) - .map((m, i) => { - const cat = - (m.metadata as any)?.category ?? - m.categories?.[0] ?? - "uncategorized"; - const imp = (m.metadata as any)?.importance ?? "?"; - const created = m.created_at ?? "unknown"; - return `${i + 1}. [${m.id}] (${cat}, importance: ${imp}, created: ${created}) ${m.memory}`; - }) - .join("\n"); - - const fullPrompt = [ - "", - dreamPrompt, - "", - "", - ``, - memoryDump, - "", - "", - "Begin consolidation. Review all memories above and execute merge, delete, and rewrite operations using the available tools.", - ].join("\n"); - - process.stdout.write(fullPrompt + "\n"); - process.stderr.write( - `Dream prompt written to stdout (${fullPrompt.length} chars). Paste it into an OpenClaw session to run consolidation.\n`, - ); - } catch (err) { - if (jsonErr(opts, `Dream failed: ${String(err)}`)) return; - console.error(`Dream failed: ${String(err)}`); - } - }); }, { descriptors: [ diff --git a/integrations/openclaw/cli/config-file.ts b/integrations/openclaw/cli/config-file.ts index 2af2f36bd..99453cb1b 100644 --- a/integrations/openclaw/cli/config-file.ts +++ b/integrations/openclaw/cli/config-file.ts @@ -189,7 +189,7 @@ export function writePluginConfigField( /** * Default skills configuration — matches configure.py output. - * Enables triage, recall (with reranking), and dream consolidation. + * Enables triage and recall with reranking. */ const DEFAULT_SKILLS_CONFIG = { triage: { enabled: true }, @@ -200,7 +200,6 @@ const DEFAULT_SKILLS_CONFIG = { keywordSearch: true, identityAlwaysInclude: true, }, - dream: { enabled: true }, domain: "companion", }; diff --git a/integrations/openclaw/dream-gate.ts b/integrations/openclaw/dream-gate.ts deleted file mode 100644 index 4d7764725..000000000 --- a/integrations/openclaw/dream-gate.ts +++ /dev/null @@ -1,214 +0,0 @@ -/** - * Dream Gate — activity tracking, gate logic, and lock mechanism - * for automatic memory consolidation. - * - * State persists in the plugin's stateDir so it survives gateway restarts. - * Lock prevents concurrent consolidation runs. - */ - -import * as path from "node:path"; -import { readText, writeText, mkdirp, unlink } from "./fs-safe.ts"; - -// ============================================================================ -// Types -// ============================================================================ - -interface DreamState { - lastConsolidatedAt: number; // ms since epoch, 0 = never - sessionsSince: number; // interactive sessions since last consolidation - lastSessionId: string | null; -} - -interface DreamLock { - pid: number; - startedAt: number; -} - -interface DreamGateConfig { - minHours: number; - minSessions: number; - minMemories: number; -} - -const DEFAULTS: DreamGateConfig = { - minHours: 24, - minSessions: 5, - minMemories: 20, -}; - -const LOCK_STALE_MS = 60 * 60 * 1000; // 1 hour - -// ============================================================================ -// State Persistence -// ============================================================================ - -function statePath(stateDir: string): string { - return path.join(stateDir, "dream-state.json"); -} - -function lockPath(stateDir: string): string { - return path.join(stateDir, "dream.lock"); -} - -function ensureDir(dir: string): void { - try { - mkdirp(dir); - } catch { - /* exists */ - } -} - -function readState(stateDir: string): DreamState { - try { - const raw = readText(statePath(stateDir)); - return JSON.parse(raw) as DreamState; - } catch { - return { lastConsolidatedAt: 0, sessionsSince: 0, lastSessionId: null }; - } -} - -function writeState(stateDir: string, state: DreamState): void { - ensureDir(stateDir); - writeText(statePath(stateDir), JSON.stringify(state, null, 2)); -} - -// ============================================================================ -// Session Tracking -// ============================================================================ - -/** - * Called from agent_end on every interactive turn. - * Increments session counter (deduped by sessionId). - */ -export function incrementSessionCount( - stateDir: string, - sessionId: string, -): void { - const state = readState(stateDir); - if (state.lastSessionId !== sessionId) { - state.sessionsSince++; - state.lastSessionId = sessionId; - writeState(stateDir, state); - } -} - -// ============================================================================ -// Gate Logic -// ============================================================================ - -/** - * Check cheap gates (time + sessions). These are local file reads only. - * Call this BEFORE any API calls. If this fails, skip the expensive - * memory count check entirely. - */ -export function checkCheapGates( - stateDir: string, - config: { minHours?: number; minSessions?: number }, -): { proceed: boolean; reason?: string } { - const minHours = config.minHours ?? DEFAULTS.minHours; - const minSessions = config.minSessions ?? DEFAULTS.minSessions; - const state = readState(stateDir); - - // Gate 1: Time (one local file read) - const hoursSince = (Date.now() - state.lastConsolidatedAt) / 3_600_000; - if (hoursSince < minHours) { - return { - proceed: false, - reason: `time: ${hoursSince.toFixed(1)}h < ${minHours}h`, - }; - } - - // Gate 2: Sessions (same file, already read) - if (state.sessionsSince < minSessions) { - return { - proceed: false, - reason: `sessions: ${state.sessionsSince} < ${minSessions}`, - }; - } - - return { proceed: true }; -} - -/** - * Check expensive memory count gate. Only call AFTER checkCheapGates passes. - */ -export function checkMemoryGate( - memoryCount: number, - config: { minMemories?: number }, -): { pass: boolean; reason?: string } { - const minMemories = config.minMemories ?? DEFAULTS.minMemories; - if (memoryCount < minMemories) { - return { pass: false, reason: `memories: ${memoryCount} < ${minMemories}` }; - } - return { pass: true }; -} - -// ============================================================================ -// Lock -// ============================================================================ - -/** - * Try to acquire the dream lock. Returns true if acquired. - * Stale locks (older than 1 hour) are reclaimed. - */ -export function acquireDreamLock(stateDir: string): boolean { - ensureDir(stateDir); - const lp = lockPath(stateDir); - - // Check existing lock - try { - const raw = readText(lp); - const lock = JSON.parse(raw) as DreamLock; - const age = Date.now() - lock.startedAt; - if (age < LOCK_STALE_MS) { - return false; // Held and not stale - } - // Stale lock — remove it before attempting exclusive create - try { - unlink(lp); - } catch { - /* race ok */ - } - } catch { - // No lock file, proceed - } - - // Atomic create with exclusive flag (wx). If two processes race, - // only one succeeds. The other gets EEXIST. - const lock: DreamLock = { pid: process.pid, startedAt: Date.now() }; - try { - writeText(lp, JSON.stringify(lock), { flag: "wx" }); - return true; - } catch { - return false; // Lost race - } -} - -/** - * Release the dream lock and record successful completion. - */ -export function releaseDreamLock(stateDir: string): void { - try { - unlink(lockPath(stateDir)); - } catch { - /* already gone */ - } -} - -/** - * Record that consolidation completed. Resets session counter. - */ -export function recordDreamCompletion(stateDir: string): void { - const state = readState(stateDir); - state.lastConsolidatedAt = Date.now(); - state.sessionsSince = 0; - state.lastSessionId = null; - writeState(stateDir, state); -} - -/** - * Get current dream state for logging/diagnostics. - */ -export function getDreamState(stateDir: string): DreamState { - return readState(stateDir); -} diff --git a/integrations/openclaw/filtering.ts b/integrations/openclaw/filtering.ts index 6c51a2798..84500fa98 100644 --- a/integrations/openclaw/filtering.ts +++ b/integrations/openclaw/filtering.ts @@ -1,6 +1,6 @@ /** * Pre-extraction message filtering: noise detection, content stripping, - * generic assistant detection, truncation, and deduplication. + * and generic assistant detection. */ import type { MemoryItem } from "./types.ts"; @@ -109,8 +109,6 @@ const NOISE_CONTENT_PATTERNS: Array<{ pattern: RegExp; replacement: string }> = }, ]; -const MAX_MESSAGE_LENGTH = 2000; - /** * Patterns indicating an assistant message is a generic acknowledgment with * no extractable facts. These are produced when the agent receives a @@ -179,19 +177,9 @@ export function stripNoiseFromContent(content: string): string { return cleaned; } -/** - * Truncate a message to `MAX_MESSAGE_LENGTH` characters, preserving the - * opening (which typically contains the summary/conclusion) and appending - * a truncation marker so the extraction model knows content was cut. - */ -function truncateMessage(content: string): string { - if (content.length <= MAX_MESSAGE_LENGTH) return content; - return content.slice(0, MAX_MESSAGE_LENGTH) + "\n[...truncated]"; -} - /** * Full pre-extraction pipeline: drop noise messages, strip noise fragments, - * filter session-specific content, and truncate remaining messages. + * and filter session-specific content without truncating remaining messages. */ export function filterMessagesForExtraction( messages: Array<{ role: string; content: string }>, @@ -206,7 +194,7 @@ export function filterMessagesForExtraction( if (isSessionSpecificContent(msg.content)) continue; const cleaned = stripNoiseFromContent(msg.content); if (!cleaned) continue; - filtered.push({ role: msg.role, content: truncateMessage(cleaned) }); + filtered.push({ role: msg.role, content: cleaned }); } return filtered; } diff --git a/integrations/openclaw/index.test.ts b/integrations/openclaw/index.test.ts index da86c5a7c..d3f4ea775 100644 --- a/integrations/openclaw/index.test.ts +++ b/integrations/openclaw/index.test.ts @@ -3,6 +3,7 @@ * message filtering logic. */ import { describe, it, expect, vi } from "vitest"; +import { createMemoryLifecycle } from "../agent-plugin-core/typescript/src/lifecycle.ts"; import memoryPlugin, { extractAgentId, effectiveUserId, @@ -521,13 +522,14 @@ What is the deployment plan?`, expect(result[0].content).toBe("What is the deployment plan?"); }); - it("truncates long messages", () => { - const longContent = "A".repeat(3000); - const messages = [{ role: "assistant", content: longContent }]; - const result = filterMessagesForExtraction(messages); - expect(result).toHaveLength(1); - expect(result[0].content.length).toBeLessThan(2100); - expect(result[0].content).toContain("[...truncated]"); + it.each(["user", "assistant"])("preserves long %s messages through extraction preparation", (role) => { + const content = "Repository detail. ".repeat(600) + "Final requirement. password=hidden-extraction-secret"; + const result = createMemoryLifecycle().prepareConversation( + filterMessagesForExtraction([{ role, content }]), + ); + expect(result).toEqual([ + { role, content: content.replace("hidden-extraction-secret", "[REDACTED]") }, + ]); }); it("returns empty array when all messages are noise", () => { diff --git a/integrations/openclaw/index.ts b/integrations/openclaw/index.ts index 58f59b6f9..835a869bd 100644 --- a/integrations/openclaw/index.ts +++ b/integrations/openclaw/index.ts @@ -43,18 +43,9 @@ import { } from "./isolation.ts"; import { loadCompactTriagePrompt, - loadDreamPrompt, isSkillsMode, } from "./skill-loader.ts"; import { recall as skillRecall, sanitizeQuery } from "./recall.ts"; -import { - incrementSessionCount, - checkCheapGates, - checkMemoryGate, - acquireDreamLock, - releaseDreamLock, - recordDreamCompletion, -} from "./dream-gate.ts"; import { PlatformBackend } from "./backend/platform.ts"; import type { Backend } from "./backend/base.ts"; import { registerCliCommands } from "./cli/commands.ts"; @@ -62,6 +53,7 @@ import { readPluginAuth } from "./cli/config-file.ts"; import { registerAllTools } from "./tools/index.ts"; import type { ToolDeps } from "./tools/index.ts"; import { captureEvent } from "./telemetry.ts"; +import { createMemoryLifecycle } from "../agent-plugin-core/typescript/src/lifecycle.ts"; import { bootstrapTelemetryFlag } from "./fs-safe.ts"; // ============================================================================ @@ -176,6 +168,8 @@ const memoryPlugin = definePluginEntry({ } const provider = createProvider(cfg, api); + const lifecycle = createMemoryLifecycle(); + lifecycle.beginSession(); // Create Backend instance — PlatformBackend for platform mode, providerToBackend adapter for OSS let backend: Backend; @@ -220,10 +214,6 @@ const memoryPlugin = definePluginEntry({ api.registerMemoryCapability({ publicArtifacts: createPublicArtifactsProvider({ provider, - cfg, - get stateDir() { - return pluginStateDir; - }, effectiveUserId: _effectiveUserId, }), runtime: { @@ -373,10 +363,10 @@ const memoryPlugin = definePluginEntry({ setCurrentSessionId: (id: string) => { currentSessionId = id; }, - getStateDir: () => pluginStateDir, }, skillsActive, _captureEvent, + lifecycle, ); // ======================================================================== @@ -420,13 +410,13 @@ function registerHooks( ) => SearchOptions, session: { setCurrentSessionId: (id: string) => void; - getStateDir: () => string | undefined; }, skillsActive: boolean = false, _captureEvent: ( event: string, props?: Record, ) => void = () => {}, + lifecycle: ReturnType = createMemoryLifecycle(), ) { // ======================================================================== // SKILLS MODE: Agentic memory via before_prompt_build @@ -493,7 +483,7 @@ function registerHooks( if (recallEnabled && recallStrategy !== "manual") { const recallStart = Date.now(); try { - const query = sanitizeQuery(event.prompt); + const query = lifecycle.prepareUserText(sanitizeQuery(event.prompt)); // Smart mode: skip session search (saves 1 API call per turn) const sessionIdForRecall = @@ -533,142 +523,18 @@ function registerHooks( ); } - // Auto-dream: check CHEAP gates first (local file reads only). - // Only hit the API for memory count if time + session gates pass. - const stateDir = session.getStateDir(); - const dreamEnabled = - cfg.skills?.dream?.enabled !== false && - cfg.skills?.dream?.auto !== false; - let dreamContext = ""; - if (dreamEnabled && stateDir && !isSubagent) { - try { - const cheapResult = checkCheapGates( - stateDir, - cfg.skills?.dream ?? {}, - ); - if (cheapResult.proceed) { - // Cheap gates passed. Now do the expensive memory count check. - const memories = await provider.getAll({ - user_id: userId, - source: "OPENCLAW", - }); - const memCount = Array.isArray(memories) ? memories.length : 0; - const memResult = checkMemoryGate( - memCount, - cfg.skills?.dream ?? {}, - ); - if (memResult.pass && acquireDreamLock(stateDir)) { - const dreamPrompt = loadDreamPrompt(cfg.skills ?? {}); - if (dreamPrompt) { - dreamContext = - "\n\n\nIMPORTANT: Before responding to the user, run a memory consolidation pass. Follow the protocol below, then respond normally.\n\n" + - dreamPrompt + - "\n"; - // Track which session triggered dream (session-keyed, not global) - dreamSessionId = sessionId; - _captureEvent("openclaw.hook.dream", { - phase: "triggered", - memory_count: memCount, - }); - api.logger.info( - `openclaw-mem0: auto-dream triggered (${memCount} memories, gate passed)`, - ); - } else { - releaseDreamLock(stateDir); - } - } - } - } catch (err) { - api.logger.warn( - `openclaw-mem0: auto-dream gate check failed: ${String(err)}`, - ); - } - } - return { prependSystemContext: systemContext, // cached by provider - prependContext: recallContext + dreamContext, // per-turn dynamic + prependContext: recallContext, // per-turn dynamic }; }); - // Session-keyed dream tracking. Only the session that triggered dream - // can complete it. Prevents cross-session false completion. - let dreamSessionId: string | undefined; - api.on("agent_end", async (event: any, ctx: any) => { const sessionId = ctx?.sessionKey ?? undefined; - const trigger = ctx?.trigger ?? undefined; if (sessionId) session.setCurrentSessionId(sessionId); - // If dream was triggered for THIS session, handle cleanup regardless - // of success/failure. A failed turn must still release the lock. - const stateDir = session.getStateDir(); - if (dreamSessionId && dreamSessionId === sessionId && stateDir) { - dreamSessionId = undefined; - - if (!event.success) { - // Turn failed/aborted after lock acquired. Release lock, do not - // record completion. Gates will re-trigger next eligible turn. - releaseDreamLock(stateDir); - api.logger.warn( - "openclaw-mem0: auto-dream turn failed, lock released, will retry", - ); - return; - } - - // Verify the model actually performed WRITE operations (not just reads). - // Only count memory_add, memory_update, memory_delete. - // Exclude memory_list and memory_search (read-only, orient-only pass). - // Scan only the LAST assistant message (this turn), not the full session - // snapshot, to avoid matching earlier tool calls from prior turns. - const WRITE_TOOLS = new Set([ - "memory_add", - "memory_update", - "memory_delete", - ]); - const messages = event.messages ?? []; - // Find the last assistant message (this turn's output) - const lastAssistant = [...messages] - .reverse() - .find((m: any) => m.role === "assistant"); - const writeToolUsed = - lastAssistant && Array.isArray(lastAssistant.content) - ? lastAssistant.content.some( - (block: any) => - block.type === "tool_use" && WRITE_TOOLS.has(block.name), - ) - : false; - - if (writeToolUsed) { - releaseDreamLock(stateDir); - recordDreamCompletion(stateDir); - _captureEvent("openclaw.hook.dream", { - phase: "completed", - write_tools_used: true, - }); - api.logger.info( - "openclaw-mem0: auto-dream completed (verified write tool usage), lock released", - ); - } else { - releaseDreamLock(stateDir); - api.logger.warn( - "openclaw-mem0: auto-dream injected but no write tools executed. Lock released, will retry.", - ); - } - return; - } - if (!event.success) return; - // Track session for dream gating (interactive turns only) - if ( - stateDir && - sessionId && - !isNonInteractiveTrigger(trigger, sessionId) - ) { - incrementSessionCount(stateDir, sessionId); - } - api.logger.info("openclaw-mem0: skills-mode agent_end (no auto-capture)"); }); @@ -733,6 +599,7 @@ function registerHooks( "", ) .trim(); + const safePrompt = lifecycle.prepareUserText(cleanPrompt); const recallStart = Date.now(); const recallWork = async () => { @@ -741,7 +608,7 @@ function registerHooks( // Search long-term memories (user-scoped; subagents read from parent namespace) let longTermResults = await provider.search( - cleanPrompt, + safePrompt, buildSearchOptions( undefined, recallTopK, @@ -767,7 +634,7 @@ function registerHooks( // Only broaden for genuinely new sessions with short prompts // (cold-start blindness). Skip on subsequent turns to save API calls. - if (isNewSession && cleanPrompt.length < 100) { + if (isNewSession && safePrompt.length < 100) { const broadOpts = buildSearchOptions( undefined, 5, @@ -1007,8 +874,9 @@ function registerHooks( content: m.content, })); - // Apply noise filtering pipeline: drop noise, strip fragments, truncate - const formattedMessages = filterMessagesForExtraction(selected); + // Filter noise and redact secrets without truncating message text. + const formattedMessages: Array<{ role: string; content: string }> = + lifecycle.prepareConversation(filterMessagesForExtraction(selected)); if (formattedMessages.length === 0) return; diff --git a/integrations/openclaw/openclaw-plugin-sdk.d.ts b/integrations/openclaw/openclaw-plugin-sdk.d.ts index 9854a5462..7eda91994 100644 --- a/integrations/openclaw/openclaw-plugin-sdk.d.ts +++ b/integrations/openclaw/openclaw-plugin-sdk.d.ts @@ -1,7 +1,7 @@ declare module "openclaw/plugin-sdk" { export interface MemoryArtifact { id: string; - type: "memory" | "dream" | "digest" | "entity"; + type: "memory" | "digest" | "entity"; title: string; content: string; metadata?: Record; diff --git a/integrations/openclaw/openclaw.plugin.json b/integrations/openclaw/openclaw.plugin.json index 34564bfe4..2d9f46518 100644 --- a/integrations/openclaw/openclaw.plugin.json +++ b/integrations/openclaw/openclaw.plugin.json @@ -2,7 +2,7 @@ "id": "openclaw-mem0", "name": "Memory (Mem0)", "description": "Mem0 memory backend for OpenClaw — platform (mem0.ai cloud) or self-hosted open-source. Auto-recall and auto-capture are opt-in (disabled by default). Supports OpenAI, Anthropic, Ollama (fully local), Qdrant, and PGVector providers.", - "version": "1.0.15", + "version": "1.1.0", "kind": "memory", "skills": ["skills"], "commandAliases": [ @@ -145,7 +145,7 @@ "skills": { "label": "Agentic Memory Skills", "advanced": true, - "help": "Enable skills-based memory extraction. The agent controls what to remember (triage), how to recall (recall), and periodic cleanup (dream). Disables auto-capture when active." + "help": "Enable skills-based memory extraction. The agent controls what to remember (triage) and how to recall (recall). Disables auto-capture when active." } }, "configSchema": { @@ -279,16 +279,6 @@ "categoryOrder": { "type": "array", "items": { "type": "string" } } } }, - "dream": { - "type": "object", - "properties": { - "enabled": { "type": "boolean" }, - "auto": { "type": "boolean" }, - "minHours": { "type": "number" }, - "minSessions": { "type": "number" }, - "minMemories": { "type": "number" } - } - }, "domain": { "type": "string" }, "customRules": { "type": "object", @@ -317,4 +307,4 @@ "hosts": ["us.i.posthog.com"] } ] -} \ No newline at end of file +} diff --git a/integrations/openclaw/package.json b/integrations/openclaw/package.json index ead4e38d0..6ea5a0aab 100644 --- a/integrations/openclaw/package.json +++ b/integrations/openclaw/package.json @@ -1,6 +1,6 @@ { "name": "@mem0/openclaw-mem0", - "version": "1.0.16", + "version": "1.1.0", "type": "module", "description": "Mem0 memory backend for OpenClaw — platform or self-hosted open-source", "license": "Apache-2.0", diff --git a/integrations/openclaw/public-artifacts.ts b/integrations/openclaw/public-artifacts.ts index a8ba105f5..cce6705ca 100644 --- a/integrations/openclaw/public-artifacts.ts +++ b/integrations/openclaw/public-artifacts.ts @@ -1,18 +1,15 @@ /** * Public Artifacts Provider for OpenClaw memory-wiki bridge mode. * - * Exposes Mem0 memories and dream state as artifacts that can be + * Exposes Mem0 memories as artifacts that can be * consumed by other plugins (e.g., memory-wiki in bridge mode). */ -import type { Mem0Provider, MemoryItem, Mem0Config } from "./types.ts"; +import type { Mem0Provider, MemoryItem } from "./types.ts"; import type { MemoryArtifact } from "openclaw/plugin-sdk"; -import { getDreamState } from "./dream-gate.ts"; export interface PublicArtifactsContext { provider: Mem0Provider; - cfg: Mem0Config; - stateDir?: string; effectiveUserId: (sessionKey?: string) => string; } @@ -28,7 +25,7 @@ export function createPublicArtifactsProvider(ctx: PublicArtifactsContext) { }): Promise { const artifacts: MemoryArtifact[] = []; const userId = options?.userId ?? ctx.effectiveUserId(); - const types = options?.types ?? ["memory", "dream", "entity"]; + const types = options?.types ?? ["memory", "entity"]; const limit = options?.limit ?? 100; try { @@ -44,14 +41,6 @@ export function createPublicArtifactsProvider(ctx: PublicArtifactsContext) { } } - // Dream state artifact (if dream enabled and stateDir available) - if (types.includes("dream") && ctx.stateDir && ctx.cfg.skills?.dream?.enabled) { - const dreamArtifact = getDreamArtifact(ctx.stateDir, userId); - if (dreamArtifact) { - artifacts.push(dreamArtifact); - } - } - // Entity artifacts (grouped memories by category) if (types.includes("entity")) { const entityArtifacts = extractEntityArtifacts(artifacts.filter(a => a.type === "memory")); @@ -90,39 +79,6 @@ function memoryToArtifact(mem: MemoryItem): MemoryArtifact { }; } -/** - * Get dream consolidation state as an artifact. - */ -function getDreamArtifact(stateDir: string, userId: string): MemoryArtifact | null { - try { - const state = getDreamState(stateDir); - if (state.lastConsolidatedAt === 0) { - return null; // No consolidation has occurred yet - } - - const lastDate = new Date(state.lastConsolidatedAt).toISOString(); - return { - id: `mem0:dream:${userId}:state`, - type: "dream", - title: `Dream State (last: ${lastDate.split("T")[0]})`, - content: [ - `Last consolidation: ${lastDate}`, - `Sessions since: ${state.sessionsSince}`, - `Last session: ${state.lastSessionId ?? "none"}`, - ].join("\n"), - metadata: { - lastConsolidatedAt: state.lastConsolidatedAt, - sessionsSince: state.sessionsSince, - lastSessionId: state.lastSessionId, - user_id: userId, - }, - updatedAt: lastDate, - }; - } catch { - return null; - } -} - /** * Extract entity artifacts from memories (grouped by category). */ diff --git a/integrations/openclaw/scripts/configure.py b/integrations/openclaw/scripts/configure.py index 26eaa673c..742a2c7b6 100755 --- a/integrations/openclaw/scripts/configure.py +++ b/integrations/openclaw/scripts/configure.py @@ -54,7 +54,6 @@ def main(): "keywordSearch": True, "identityAlwaysInclude": True, }, - "dream": {"enabled": True}, "domain": "companion", }, }, diff --git a/integrations/openclaw/skill-loader.ts b/integrations/openclaw/skill-loader.ts index 628c98e23..7e7fbce94 100644 --- a/integrations/openclaw/skill-loader.ts +++ b/integrations/openclaw/skill-loader.ts @@ -346,12 +346,12 @@ export function loadSkill( } // Inject triage knobs (importanceThreshold, credentialPatterns) - if (skillName === "memory-triage" || skillName === "memory-dream") { + if (skillName === "memory-triage") { const knobs = renderTriageKnobs(config); if (knobs) parts.push(knobs); } - // Append user custom rules (triage-only — extraction rules don't apply to recall/dream) + // Append user custom rules to the triage prompt. if (skillName === "memory-triage" && config.customRules) { const rulesBlock: string[] = ["\n## User Custom Rules\n"]; if (config.customRules.include?.length) { @@ -659,15 +659,6 @@ export function loadCompactTriagePrompt(config: SkillsConfig = {}): string { return parts.join("\n"); } -/** - * Load the dream skill prompt for consolidation sessions. - */ -export function loadDreamPrompt(config: SkillsConfig = {}): string { - const dream = loadSkill("memory-dream", config); - if (!dream) return ""; - return dream.prompt; -} - /** * Resolve the effective categories — user overrides merged with defaults. */ diff --git a/integrations/openclaw/skills/memory-dream/SKILL.md b/integrations/openclaw/skills/memory-dream/SKILL.md deleted file mode 100644 index e6ccf0032..000000000 --- a/integrations/openclaw/skills/memory-dream/SKILL.md +++ /dev/null @@ -1,150 +0,0 @@ ---- -name: memory-dream -description: > - Memory consolidation protocol. Reviews all stored memories, merges duplicates, - removes noise and credentials, rewrites unclear entries, and enforces TTL expiration. - Use when the user asks to clean up, consolidate, or review their memories. - Also triggers automatically after sufficient activity (configurable). -user-invocable: true -metadata: - {"openclaw": {"injected": true, "emoji": "💤", "requires": {"env": ["MEM0_API_KEY", "OPENAI_API_KEY", "ANTHROPIC_API_KEY"], "bins": []}}} ---- - -# Memory Consolidation - -You are performing a memory consolidation pass. Your goal is to review all stored memories for this user and improve their overall quality. Think of this as compressing raw observations into clean, durable knowledge. - -## Available Tools - -### memory_search -Semantic search across stored memories. -- `query` (required): search query -- `limit`: max results -- `userId`, `agentId`: scope overrides -- `scope`: `"all"` (default), `"session"`, or `"long-term"` -- `categories`: filter by category array - -### memory_add -Store new facts in long-term memory. -- `facts` (required): array of facts — ALL must share the same category -- `category`: `"identity"`, `"preference"`, `"decision"`, `"rule"`, `"project"`, `"configuration"`, `"technical"`, `"relationship"` -- `importance`: 0.0–1.0 - -### memory_get -Retrieve a single memory by ID. -- `memoryId` (required): the memory ID - -### memory_list -List all stored memories for a user or agent. -- `userId`, `agentId`: scope overrides -- `scope`: `"all"` (default), `"session"`, or `"long-term"` - -### memory_update -Update an existing memory's text in place. Atomic and preserves edit history. -- `memoryId` (required): the memory ID to update -- `text` (required): the new text (replaces old) - -### memory_delete -Delete memories by ID, query, or bulk. -- `memoryId`: specific memory ID to delete -- `all`: delete ALL memories (requires `confirm: true`) -- `userId`, `agentId`: scope overrides - -### memory_event_list -List recent background processing events (platform mode only). - -### memory_event_status -Get status of a specific background event. -- `event_id` (required): the event ID to check - -Follow these four phases in order. Do not skip phases. - -## Phase 1: Orient - -Survey the current memory landscape before making any changes. - -1. Call `memory_list` to load all stored memories. -2. Count memories by category. Note the total. -3. Identify the oldest and newest memories by their timestamps. -4. Note any obvious problems visible in the list: duplicates, very short entries, entries without temporal anchors. - -Do not modify anything in this phase. The goal is to understand what you are working with. - -## Phase 2: Gather Targets - -Identify which memories need action. Use the tools to investigate. - -**Search for recent additions:** -Call `memory_search` with a `created_at` filter to find memories added since the last consolidation. These are the most likely to need merging or cleanup. - -**Classify each target into one of these actions:** -- DELETE: contains credentials, expired by TTL, pure noise, raw tool output, standalone timestamps -- MERGE: two or more memories express the same fact in different words, or a series tracks incremental changes to the same entity -- REWRITE: vague, missing temporal anchor, uses first person instead of third, wrong category, overly verbose - -## Phase 3: Consolidate - -Execute the actions identified in Phase 2. Work in this priority order: - -### 3a. Delete dangerous and expired entries - -Delete immediately using `memory_delete`: -- Credentials, API keys, tokens, passwords, secrets (matching known credential prefixes and auth patterns injected by the plugin at runtime) -- Pure timestamps with no context -- Raw tool output stored as memory -- Heartbeat or cron execution records -- Generic acknowledgments stored as memory ("ok", "got it") -- Operational memories older than 7 days -- Project memories older than 90 days - -### 3b. Merge duplicates - -When two or more memories express the same fact: -1. Pick the most complete version as the base -2. Call `memory_update` on the best version to incorporate missing details from the others -3. Call `memory_delete` on the redundant entries - -`memory_update` is preferred over forget-then-store because it is atomic and preserves edit history. - -When merging, follow these rules: -- Keep the user's original words for opinions and preferences -- Preserve temporal anchors from both versions -- Do not exceed 50 words in the merged result -- The merged memory must be self-contained (understandable without the deleted ones) - -### 3c. Rewrite unclear entries - -When a memory needs improvement but is not a duplicate: -1. Call `memory_update` with the improved text - -Rewrite when: -- Memory uses first person ("I prefer") instead of third ("User prefers") -- Memory lacks a temporal anchor for time-sensitive information -- Memory is vague ("likes python") and can be made specific ("User prefers Python for backend development") -- Memory has the wrong category assignment -- Memory is over 50 words and can be compressed without losing information - -## Phase 4: Report - -After completing all operations, summarize what you did: - -``` -Consolidation complete. -- Reviewed: [total count] -- Deleted (credentials/secrets): [count] -- Deleted (expired/stale): [count] -- Merged: [count] groups into [count] memories -- Rewritten: [count] -- Final count: [total remaining] -- Issues found: [any notable problems or observations] -``` - -## Quality Targets - -After consolidation, the memory store should have: -- Zero memories containing credentials or secrets -- Zero duplicate memories (same fact in different words) -- All project and operational memories have temporal anchors ("As of YYYY-MM-DD") -- All memories use third person voice -- All memories are correctly categorized -- Each memory is 15-50 words, self-contained, and atomic (one fact per memory) diff --git a/integrations/openclaw/telemetry.ts b/integrations/openclaw/telemetry.ts index 3e2b4c7a4..fc1c6b99d 100644 --- a/integrations/openclaw/telemetry.ts +++ b/integrations/openclaw/telemetry.ts @@ -1,301 +1,115 @@ -/** - * Plugin telemetry — anonymous usage tracking via PostHog. - * - * Sends fire-and-forget events to PostHog using native fetch(). - * Events are batched and flushed every 5 seconds or when the queue - * reaches 10 events, whichever comes first. - * - * Disable with: MEM0_TELEMETRY=false - */ - import { createHash, randomUUID } from "node:crypto"; -import { readPluginAuth, writePluginAuth, getBaseUrl, clearAnonymousTelemetryId } from "./cli/config-file.ts"; + +import { createTelemetry } from "../agent-plugin-core/typescript/src/telemetry.ts"; +import { clearAnonymousTelemetryId, getBaseUrl, readPluginAuth, writePluginAuth } from "./cli/config-file.ts"; declare const __OPENCLAW_PLUGIN_VERSION__: string; export const PLUGIN_VERSION: string = __OPENCLAW_PLUGIN_VERSION__; -const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"; -const POSTHOG_HOST = "https://us.i.posthog.com/i/v0/e/"; +let cachedAnonymousId: string | undefined; +let aliasCheckDone = false; +let emailResolutionAttempted = false; +let currentDistinctId = ""; -const FLUSH_INTERVAL_MS = 5_000; -const FLUSH_THRESHOLD = 10; - -let eventQueue: Record[] = []; -let flushTimer: ReturnType | undefined; - -let _cachedAnonymousId: string | undefined; -let _aliasCheckDone = false; - -/** - * Return a persistent per-machine anonymous ID, generating one if needed. - * - * Stored in ~/.openclaw/openclaw.json under the plugin's `anonymousTelemetryId` - * field so repeat sessions on the same machine share one PostHog identity - * instead of collapsing into a single shared fallback string. The result is - * cached in module memory after the first read so we don't re-touch disk on - * every queued event. - */ -function getOrCreateAnonymousId(): string { - if (_cachedAnonymousId) return _cachedAnonymousId; - try { - const auth = readPluginAuth(); - if (auth.anonymousTelemetryId) { - _cachedAnonymousId = auth.anonymousTelemetryId; - return _cachedAnonymousId; - } - } catch { - /* ignore */ - } - const newId = `openclaw-anon-${randomUUID().replace(/-/g, "")}`; - try { - writePluginAuth({ anonymousTelemetryId: newId }); - } catch { - /* ignore — return generated id anyway */ - } - _cachedAnonymousId = newId; - return newId; +function enabled(): boolean { + const value = (globalThis as any).__mem0_telemetry_override ?? process.env.MEM0_TELEMETRY; + return value === undefined || !["false", "0", "no", "off"].includes(String(value).toLowerCase()); } -/** - * If we just resolved to a real identity but a stored anonymous id exists, - * build a one-shot PostHog $identify event so the pre-signup history gets - * stitched onto the authenticated profile. Returns null when no aliasing is - * needed (already done, or no anon id on disk, or still anonymous). - * - * Caller is responsible for pushing the returned event onto eventQueue ahead - * of the regular event. - */ -function maybeBuildIdentifyEvent( - distinctId: string, -): Record | null { - if (_aliasCheckDone) return null; - if (!distinctId || distinctId.startsWith("openclaw-anon-")) return null; +function anonymousId(): string { + if (cachedAnonymousId) return cachedAnonymousId; try { - const auth = readPluginAuth(); - const storedAnon = auth.anonymousTelemetryId; - if (!storedAnon) { - _aliasCheckDone = true; - return null; - } - const identifyEvent = { - event: "$identify", - distinct_id: distinctId, - properties: { - $anon_distinct_id: storedAnon, - $lib: "posthog-node", - }, - }; - // Clear the anonymous ID from config after aliasing (don't write empty string) - try { - clearAnonymousTelemetryId(); - } catch { - /* ignore — alias may double-fire next session, harmless */ - } - _aliasCheckDone = true; - _cachedAnonymousId = undefined; - return identifyEvent; + const stored = readPluginAuth().anonymousTelemetryId; + if (stored) return (cachedAnonymousId = stored); } catch { - return null; + // First run or unreadable config. + } + const created = `openclaw-anon-${randomUUID().replace(/-/g, "")}`; + try { + writePluginAuth({ anonymousTelemetryId: created }); + } catch { + // An unwritable config must not break the plugin. + } + return (cachedAnonymousId = created); +} + +function distinctId(apiKey?: string): string { + try { + const email = readPluginAuth().userEmail; + if (email) return createHash("sha256").update(email).digest("hex"); + } catch { + // Fall through to the API key or anonymous identity. + } + return apiKey ? createHash("sha256").update(apiKey).digest("hex") : anonymousId(); +} + +const telemetry = createTelemetry({ + host: "openclaw", + source: "OPENCLAW", + version: PLUGIN_VERSION, + distinctId: () => currentDistinctId, + enabled, +}); + +function identifyAnonymous(id: string): void { + if (aliasCheckDone || id.startsWith("openclaw-anon-")) return; + try { + const anonymous = readPluginAuth().anonymousTelemetryId; + aliasCheckDone = true; + if (!anonymous) return; + telemetry.capture("$identify", { $anon_distinct_id: anonymous }); + clearAnonymousTelemetryId(); + cachedAnonymousId = undefined; + } catch { + // Aliasing is best effort. } } -let _emailResolutionAttempted = false; - -/** - * If we have an apiKey but no cached userEmail, do a one-shot /v1/ping/ - * call to resolve the email and cache it. This runs async as a side-effect; - * the current event ships with md5(apiKey) but subsequent events (including - * those flushed by the beforeExit handler in the same process) will use - * the resolved email. - */ -function maybeResolveEmail(apiKey: string): void { - if (_emailResolutionAttempted) return; - _emailResolutionAttempted = true; - - const baseUrl = getBaseUrl().replace(/\/+$/, ""); - fetch(`${baseUrl}/v1/ping/`, { +function resolveEmail(apiKey: string): void { + if (emailResolutionAttempted) return; + emailResolutionAttempted = true; + fetch(`${getBaseUrl().replace(/\/+$/, "")}/v1/ping/`, { method: "GET", - headers: { - Authorization: `Token ${apiKey}`, - "Content-Type": "application/json", - }, + headers: { Authorization: `Token ${apiKey}`, "Content-Type": "application/json" }, signal: AbortSignal.timeout(5_000), }) - .then((res) => res.json()) + .then((response) => response.json()) .then((data: any) => { - const email = data?.user_email; - if (email) { - try { - writePluginAuth({ userEmail: email }); - } catch { - /* ignore */ - } - const oldId = createHash("sha256").update(apiKey).digest("hex"); - const newId = createHash("sha256").update(email).digest("hex"); - for (const ev of eventQueue) { - if (ev.distinct_id === oldId) { - ev.distinct_id = newId; - } - } + if (!data?.user_email) return; + writePluginAuth({ userEmail: data.user_email }); + const oldId = createHash("sha256").update(apiKey).digest("hex"); + const newId = createHash("sha256").update(data.user_email).digest("hex"); + for (const event of telemetry.queueForTesting()) { + if (event.distinct_id === oldId) event.distinct_id = newId; } }) .catch(() => { - /* silently swallow — md5(apiKey) is used as fallback */ + // The API-key hash remains a stable fallback. }); } -let _telemetryEnabled: boolean | undefined; -function isTelemetryEnabled(): boolean { - if (_telemetryEnabled !== undefined) return _telemetryEnabled; - try { - const val = (globalThis as any).__mem0_telemetry_override; - if (val !== undefined) { - const s = String(val).toLowerCase(); - _telemetryEnabled = s !== "false" && s !== "0" && s !== "no"; - } else { - _telemetryEnabled = true; - } - } catch { - _telemetryEnabled = true; - } - return _telemetryEnabled; -} - -/** - * Return a stable anonymous identifier for the current user. - * - * Priority: cached userEmail (from /v1/ping/) > MD5(apiKey) > - * persistent per-machine anonymous ID. - */ -function getDistinctId(apiKey?: string): string { - try { - const auth = readPluginAuth(); - if (auth.userEmail) { - return createHash("sha256").update(auth.userEmail).digest("hex"); - } - } catch { - /* ignore */ - } - if (apiKey) { - return createHash("sha256").update(apiKey).digest("hex"); - } - return getOrCreateAnonymousId(); -} - -function ensureFlushTimer(): void { - if (flushTimer) return; - flushTimer = setInterval(flushEvents, FLUSH_INTERVAL_MS); - if (typeof flushTimer === "object" && "unref" in flushTimer) { - flushTimer.unref(); - } -} - -let _exitHandlerInstalled = false; - -/** - * Install a one-time `beforeExit` handler that drains queued events on - * process exit. Without this, short-lived CLI invocations (e.g. one - * `openclaw mem0 status` call) exit before the unref'd flushTimer fires - * and before FLUSH_THRESHOLD is hit, dropping every queued event silently. - * - * Returning a Promise from a `beforeExit` handler keeps the event loop - * alive until that Promise resolves, so the awaited fetch actually has - * time to land at PostHog. - */ -function ensureExitHandler(): void { - if (_exitHandlerInstalled) return; - _exitHandlerInstalled = true; - process.on("beforeExit", async () => { - if (eventQueue.length === 0) return; - const batch = eventQueue; - eventQueue = []; - const body = JSON.stringify({ api_key: POSTHOG_API_KEY, batch }); - try { - await fetch(POSTHOG_HOST, { - method: "POST", - headers: { - "Content-Type": "application/json", - "Content-Length": String(Buffer.byteLength(body)), - }, - body, - signal: AbortSignal.timeout(3_000), - }); - } catch { - /* silently swallow */ - } - }); -} - -function flushEvents(): void { - if (eventQueue.length === 0) return; - const batch = eventQueue; - eventQueue = []; - - const body = JSON.stringify({ api_key: POSTHOG_API_KEY, batch }); - fetch(POSTHOG_HOST, { - method: "POST", - headers: { - "Content-Type": "application/json", - "Content-Length": String(Buffer.byteLength(body)), - }, - body, - signal: AbortSignal.timeout(3_000), - }).catch(() => { - /* silently swallow */ - }); -} - -/** - * Capture a PostHog event (non-blocking, never throws). - */ export function captureEvent( eventName: string, properties: Record = {}, - ctx?: { apiKey?: string; mode?: string; skillsActive?: boolean }, + context?: { apiKey?: string; mode?: string; skillsActive?: boolean }, ): void { - if (!isTelemetryEnabled()) return; - + if (!enabled()) return; try { - const distinctId = getDistinctId(ctx?.apiKey); - + currentDistinctId = distinctId(context?.apiKey); let hasEmail = false; - try { hasEmail = !!readPluginAuth().userEmail; } catch { /* ignore */ } - if (ctx?.apiKey && !hasEmail && !distinctId.startsWith("openclaw-anon-")) { - maybeResolveEmail(ctx.apiKey); + try { + hasEmail = Boolean(readPluginAuth().userEmail); + } catch { + // Resolve it below when possible. } - - // First authenticated event after a previous anonymous session: queue a - // $identify ahead of the regular event so PostHog merges the anonymous - // history onto the authenticated profile in the same batch flush. - const identifyEvent = maybeBuildIdentifyEvent(distinctId); - if (identifyEvent) { - eventQueue.push(identifyEvent); - } - - eventQueue.push({ - event: eventName, - distinct_id: distinctId, - properties: { - source: "OPENCLAW", - language: "node", - plugin_version: PLUGIN_VERSION, - node_version: process.version, - os: process.platform, - mode: ctx?.mode, - skills_active: ctx?.skillsActive, - $process_person_profile: false, - $lib: "posthog-node", - ...properties, - }, + if (context?.apiKey && !hasEmail) resolveEmail(context.apiKey); + identifyAnonymous(currentDistinctId); + telemetry.capture(eventName, { + mode: context?.mode, + skills_active: context?.skillsActive, + ...properties, }); - - ensureFlushTimer(); - ensureExitHandler(); - - if (eventQueue.length >= FLUSH_THRESHOLD) { - flushEvents(); - } } catch { - /* silently swallow */ + // Telemetry must never affect plugin behavior. } } diff --git a/integrations/openclaw/tests/cli-commands.test.ts b/integrations/openclaw/tests/cli-commands.test.ts index 77cd52f49..ed036e9c7 100644 --- a/integrations/openclaw/tests/cli-commands.test.ts +++ b/integrations/openclaw/tests/cli-commands.test.ts @@ -29,11 +29,6 @@ vi.mock("../fs-safe.ts", () => ({ unlink: vi.fn(), })); -vi.mock("../skill-loader.ts", () => ({ - loadDreamPrompt: vi.fn().mockReturnValue("dream prompt"), -})); - - // --------------------------------------------------------------------------- // Imports (after mocks) // --------------------------------------------------------------------------- @@ -46,7 +41,6 @@ import { enableSkillsConfig, getBaseUrl, } from "../cli/config-file.ts"; -import { loadDreamPrompt } from "../skill-loader.ts"; // --------------------------------------------------------------------------- // Mock Commander program builder @@ -263,8 +257,6 @@ describe("registerCliCommands", () => { (readPluginAuth as ReturnType).mockReturnValue({}); (writePluginAuth as ReturnType).mockImplementation(() => {}); (getBaseUrl as ReturnType).mockReturnValue("https://api.mem0.ai"); - (loadDreamPrompt as ReturnType).mockReturnValue("dream prompt"); - consoleSpy = { log: vi.spyOn(console, "log").mockImplementation(() => {}), error: vi.spyOn(console, "error").mockImplementation(() => {}), @@ -311,7 +303,6 @@ describe("registerCliCommands", () => { expect(names).toContain("delete"); expect(names).toContain("status"); expect(names).toContain("config"); - expect(names).toContain("dream"); }); it("registers config subcommands: show, get, set", () => { @@ -1037,6 +1028,19 @@ describe("registerCliCommands", () => { // ======================================================================== describe("status subcommand", () => { + it("reports setup instructions when Mem0 is not configured", async () => { + const { mem0, backend, cfg } = setup(); + (cfg as any).needsSetup = true; + const statusCmd = findCommand(mem0, "status")!; + + await statusCmd._action!(); + + expect(backend.status).not.toHaveBeenCalled(); + expect(consoleSpy.log).toHaveBeenCalledWith( + expect.stringContaining("openclaw mem0 init"), + ); + }); + it("calls backend.status and prints connection info", async () => { const { mem0, backend } = setup(); const statusCmd = findCommand(mem0, "status")!; @@ -1280,104 +1284,6 @@ describe("registerCliCommands", () => { }); }); - // ======================================================================== - // dream subcommand - // ======================================================================== - - describe("dream subcommand", () => { - it("fetches memories and outputs dream prompt to stdout", async () => { - const { mem0, provider } = setup(); - provider.getAll.mockResolvedValueOnce([ - { - id: "m1", - memory: "User is an engineer", - categories: ["identity"], - metadata: { category: "identity", importance: 0.9 }, - created_at: "2026-01-01", - }, - ]); - const stdoutSpy = vi.spyOn(process.stdout, "write").mockImplementation(() => true); - const dreamCmd = findCommand(mem0, "dream")!; - - await dreamCmd._action!({}); - - expect(provider.getAll).toHaveBeenCalledWith( - expect.objectContaining({ - user_id: "testuser", - source: "OPENCLAW", - }), - ); - expect(loadDreamPrompt).toHaveBeenCalled(); - - // stdout should contain the dream prompt - const stdoutOutput = stdoutSpy.mock.calls.map((c) => c[0]).join(""); - expect(stdoutOutput).toContain(""); - expect(stdoutOutput).toContain("dream prompt"); - expect(stdoutOutput).toContain(" { - const { mem0, provider } = setup(); - provider.getAll.mockResolvedValueOnce([ - { id: "m1", memory: "test", categories: [], metadata: {}, created_at: "2026-01-01" }, - ]); - const stdoutSpy = vi.spyOn(process.stdout, "write").mockImplementation(() => true); - const dreamCmd = findCommand(mem0, "dream")!; - - await dreamCmd._action!({ dryRun: true }); - - // Dry run should write inventory to stderr, NOT dream prompt to stdout - expect(stderrSpy).toHaveBeenCalledWith( - expect.stringContaining("Dry run"), - ); - expect(stdoutSpy).not.toHaveBeenCalled(); - - stdoutSpy.mockRestore(); - }); - - it("prints message when no memories to consolidate", async () => { - const { mem0, provider } = setup(); - provider.getAll.mockResolvedValueOnce([]); - const dreamCmd = findCommand(mem0, "dream")!; - - await dreamCmd._action!({}); - - expect(consoleSpy.log).toHaveBeenCalledWith( - "No memories to consolidate.", - ); - }); - - it("prints error when dream skill file is not found", async () => { - const { mem0, provider } = setup(); - provider.getAll.mockResolvedValueOnce([ - { id: "m1", memory: "test", categories: [], metadata: {}, created_at: "2026-01-01" }, - ]); - (loadDreamPrompt as ReturnType).mockReturnValueOnce(""); - const dreamCmd = findCommand(mem0, "dream")!; - - await dreamCmd._action!({}); - - expect(stderrSpy).toHaveBeenCalledWith( - expect.stringContaining("Dream skill file not found"), - ); - }); - - it("handles dream errors gracefully", async () => { - const { mem0, provider } = setup(); - provider.getAll.mockRejectedValueOnce(new Error("dream boom")); - const dreamCmd = findCommand(mem0, "dream")!; - - await dreamCmd._action!({}); - - expect(consoleSpy.error).toHaveBeenCalledWith( - expect.stringContaining("Dream failed"), - ); - }); - }); - // ======================================================================== // import subcommand // ======================================================================== @@ -1647,7 +1553,7 @@ describe("registerCliCommands", () => { // ======================================================================== describe("--json flag registration", () => { - for (const name of ["search", "add", "get", "list", "update", "delete", "status", "import", "dream"]) { + for (const name of ["search", "add", "get", "list", "update", "delete", "status", "import"]) { it(`registers --json on ${name}`, () => { const { mem0 } = setup(); const cmd = findCommand(mem0, name)!; diff --git a/integrations/openclaw/tests/config.test.ts b/integrations/openclaw/tests/config.test.ts index 6eb7441ac..2c6dbac2e 100644 --- a/integrations/openclaw/tests/config.test.ts +++ b/integrations/openclaw/tests/config.test.ts @@ -1,6 +1,8 @@ /** * Tests for config.ts — mem0ConfigSchema.parse() and exported constants. */ +import { readFileSync } from "node:fs"; + import { describe, it, expect } from "vitest"; import { mem0ConfigSchema, @@ -8,6 +10,14 @@ import { DEFAULT_CUSTOM_CATEGORIES, } from "../config.ts"; +describe("plugin manifest", () => { + it("matches the package version", () => { + const packageJson = JSON.parse(readFileSync(new URL("../package.json", import.meta.url), "utf8")); + const manifest = JSON.parse(readFileSync(new URL("../openclaw.plugin.json", import.meta.url), "utf8")); + expect(manifest.version).toBe(packageJson.version); + }); +}); + // --------------------------------------------------------------------------- // Exported constants // --------------------------------------------------------------------------- @@ -379,13 +389,6 @@ describe("mem0ConfigSchema.parse() — skills config", () => { tokenBudget: 2000, maxMemories: 10, }, - dream: { - enabled: true, - auto: true, - minHours: 12, - minSessions: 3, - minMemories: 15, - }, domain: "engineering", customRules: { include: ["tool configs"], diff --git a/integrations/openclaw/tests/dream-gate.test.ts b/integrations/openclaw/tests/dream-gate.test.ts deleted file mode 100644 index cf8dc26db..000000000 --- a/integrations/openclaw/tests/dream-gate.test.ts +++ /dev/null @@ -1,376 +0,0 @@ -/** - * Tests for dream-gate.ts — activity tracking, gate logic, and lock mechanism - * for automatic memory consolidation. - * - * All filesystem operations are mocked via fs-safe.ts. - * Time-dependent tests use vi.useFakeTimers(). - */ -import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; - -vi.mock("../fs-safe.ts", () => ({ - readText: vi.fn(), - writeText: vi.fn(), - mkdirp: vi.fn(), - unlink: vi.fn(), -})); - -import { readText, writeText, mkdirp, unlink } from "../fs-safe.ts"; -import { - incrementSessionCount, - checkCheapGates, - checkMemoryGate, - acquireDreamLock, - releaseDreamLock, - recordDreamCompletion, - getDreamState, -} from "../dream-gate.ts"; - -// --------------------------------------------------------------------------- -// Helpers -// --------------------------------------------------------------------------- - -const mockReadText = readText as ReturnType; -const mockWriteText = writeText as ReturnType; -const mockMkdirp = mkdirp as ReturnType; -const mockUnlink = unlink as ReturnType; - -const STATE_DIR = "/tmp/test-state"; - -interface DreamState { - lastConsolidatedAt: number; - sessionsSince: number; - lastSessionId: string | null; -} - -function setDreamState(state: DreamState): void { - mockReadText.mockImplementation((filePath: string) => { - if (filePath.endsWith("dream-state.json")) { - return JSON.stringify(state); - } - throw new Error("ENOENT"); - }); -} - -function setNoState(): void { - mockReadText.mockImplementation(() => { - throw new Error("ENOENT"); - }); -} - -function getWrittenState(): DreamState { - const call = mockWriteText.mock.calls.find((c: unknown[]) => - (c[0] as string).endsWith("dream-state.json"), - ); - if (!call) throw new Error("No state file written"); - return JSON.parse(call[1] as string); -} - -beforeEach(() => { - vi.resetAllMocks(); - mockMkdirp.mockReturnValue(undefined); - mockUnlink.mockReturnValue(undefined); -}); - -// --------------------------------------------------------------------------- -// incrementSessionCount -// --------------------------------------------------------------------------- - -describe("incrementSessionCount", () => { - it("increments counter for a new session", () => { - setDreamState({ - lastConsolidatedAt: 0, - sessionsSince: 3, - lastSessionId: "session-old", - }); - - incrementSessionCount(STATE_DIR, "session-new"); - - const written = getWrittenState(); - expect(written.sessionsSince).toBe(4); - expect(written.lastSessionId).toBe("session-new"); - }); - - it("deduplicates same session (no increment)", () => { - setDreamState({ - lastConsolidatedAt: 0, - sessionsSince: 3, - lastSessionId: "session-same", - }); - - incrementSessionCount(STATE_DIR, "session-same"); - - // writeText should NOT have been called for the state file - const stateWrites = mockWriteText.mock.calls.filter((c: unknown[]) => - (c[0] as string).endsWith("dream-state.json"), - ); - expect(stateWrites).toHaveLength(0); - }); -}); - -// --------------------------------------------------------------------------- -// checkCheapGates -// --------------------------------------------------------------------------- - -describe("checkCheapGates", () => { - beforeEach(() => { - vi.useFakeTimers(); - }); - - afterEach(() => { - vi.useRealTimers(); - }); - - it("fails time gate when consolidation was too recent", () => { - const now = Date.now(); - vi.setSystemTime(now); - - // Last consolidated 1 hour ago, but minHours is 24 - setDreamState({ - lastConsolidatedAt: now - 1 * 3_600_000, - sessionsSince: 100, - lastSessionId: null, - }); - - const result = checkCheapGates(STATE_DIR, { minHours: 24, minSessions: 5 }); - expect(result.proceed).toBe(false); - expect(result.reason).toContain("time"); - }); - - it("fails session gate when too few sessions", () => { - const now = Date.now(); - vi.setSystemTime(now); - - // Last consolidated 48 hours ago (passes time gate), but only 2 sessions - setDreamState({ - lastConsolidatedAt: now - 48 * 3_600_000, - sessionsSince: 2, - lastSessionId: null, - }); - - const result = checkCheapGates(STATE_DIR, { - minHours: 24, - minSessions: 5, - }); - expect(result.proceed).toBe(false); - expect(result.reason).toContain("sessions"); - }); - - it("passes both gates when conditions are met", () => { - const now = Date.now(); - vi.setSystemTime(now); - - // 48 hours ago, 10 sessions — both gates pass - setDreamState({ - lastConsolidatedAt: now - 48 * 3_600_000, - sessionsSince: 10, - lastSessionId: null, - }); - - const result = checkCheapGates(STATE_DIR, { - minHours: 24, - minSessions: 5, - }); - expect(result.proceed).toBe(true); - expect(result.reason).toBeUndefined(); - }); - - it("uses defaults when config is empty", () => { - const now = Date.now(); - vi.setSystemTime(now); - - // Never consolidated (0), 100 sessions — should pass with defaults (24h, 5 sessions) - setDreamState({ - lastConsolidatedAt: 0, - sessionsSince: 100, - lastSessionId: null, - }); - - const result = checkCheapGates(STATE_DIR, {}); - expect(result.proceed).toBe(true); - }); -}); - -// --------------------------------------------------------------------------- -// checkMemoryGate -// --------------------------------------------------------------------------- - -describe("checkMemoryGate", () => { - it("fails when too few memories", () => { - const result = checkMemoryGate(5, { minMemories: 20 }); - expect(result.pass).toBe(false); - expect(result.reason).toContain("memories"); - expect(result.reason).toContain("5"); - }); - - it("passes when enough memories", () => { - const result = checkMemoryGate(25, { minMemories: 20 }); - expect(result.pass).toBe(true); - expect(result.reason).toBeUndefined(); - }); -}); - -// --------------------------------------------------------------------------- -// acquireDreamLock -// --------------------------------------------------------------------------- - -describe("acquireDreamLock", () => { - beforeEach(() => { - vi.useFakeTimers(); - }); - - afterEach(() => { - vi.useRealTimers(); - }); - - it("succeeds when no lock exists", () => { - const now = Date.now(); - vi.setSystemTime(now); - - // readText throws for lock file (not found), writeText succeeds for wx create - mockReadText.mockImplementation(() => { - throw new Error("ENOENT"); - }); - mockWriteText.mockReturnValue(undefined); - - const result = acquireDreamLock(STATE_DIR); - expect(result).toBe(true); - - // Verify it wrote a lock file with wx flag - const lockWrite = mockWriteText.mock.calls.find((c: unknown[]) => - (c[0] as string).endsWith("dream.lock"), - ); - expect(lockWrite).toBeDefined(); - const lockData = JSON.parse(lockWrite![1] as string); - expect(lockData.pid).toBe(process.pid); - expect(lockData.startedAt).toBe(now); - expect(lockWrite![2]).toEqual({ flag: "wx" }); - }); - - it("fails when lock exists and is fresh", () => { - const now = Date.now(); - vi.setSystemTime(now); - - // Lock was created 10 minutes ago — still fresh (< 1 hour) - mockReadText.mockImplementation((filePath: string) => { - if (filePath.endsWith("dream.lock")) { - return JSON.stringify({ - pid: 12345, - startedAt: now - 10 * 60 * 1000, - }); - } - throw new Error("ENOENT"); - }); - - const result = acquireDreamLock(STATE_DIR); - expect(result).toBe(false); - - // Should NOT have written a new lock - const lockWrites = mockWriteText.mock.calls.filter((c: unknown[]) => - (c[0] as string).endsWith("dream.lock"), - ); - expect(lockWrites).toHaveLength(0); - }); - - it("succeeds when lock is stale (>1hr old)", () => { - const now = Date.now(); - vi.setSystemTime(now); - - // Lock was created 2 hours ago — stale - mockReadText.mockImplementation((filePath: string) => { - if (filePath.endsWith("dream.lock")) { - return JSON.stringify({ - pid: 99999, - startedAt: now - 2 * 60 * 60 * 1000, - }); - } - throw new Error("ENOENT"); - }); - mockWriteText.mockReturnValue(undefined); - - const result = acquireDreamLock(STATE_DIR); - expect(result).toBe(true); - - // Should have unlinked the stale lock - expect(mockUnlink).toHaveBeenCalled(); - - // Should have written a new lock - const lockWrite = mockWriteText.mock.calls.find((c: unknown[]) => - (c[0] as string).endsWith("dream.lock"), - ); - expect(lockWrite).toBeDefined(); - }); -}); - -// --------------------------------------------------------------------------- -// releaseDreamLock -// --------------------------------------------------------------------------- - -describe("releaseDreamLock", () => { - it("removes lock file", () => { - releaseDreamLock(STATE_DIR); - expect(mockUnlink).toHaveBeenCalledWith( - expect.stringContaining("dream.lock"), - ); - }); -}); - -// --------------------------------------------------------------------------- -// recordDreamCompletion -// --------------------------------------------------------------------------- - -describe("recordDreamCompletion", () => { - beforeEach(() => { - vi.useFakeTimers(); - }); - - afterEach(() => { - vi.useRealTimers(); - }); - - it("resets session counter and records timestamp", () => { - const now = 1700000000000; - vi.setSystemTime(now); - - setDreamState({ - lastConsolidatedAt: 0, - sessionsSince: 15, - lastSessionId: "session-xyz", - }); - - recordDreamCompletion(STATE_DIR); - - const written = getWrittenState(); - expect(written.lastConsolidatedAt).toBe(now); - expect(written.sessionsSince).toBe(0); - expect(written.lastSessionId).toBeNull(); - }); -}); - -// --------------------------------------------------------------------------- -// getDreamState -// --------------------------------------------------------------------------- - -describe("getDreamState", () => { - it("returns default state when no file exists", () => { - setNoState(); - - const state = getDreamState(STATE_DIR); - expect(state).toEqual({ - lastConsolidatedAt: 0, - sessionsSince: 0, - lastSessionId: null, - }); - }); - - it("returns persisted state when file exists", () => { - const persisted = { - lastConsolidatedAt: 1700000000000, - sessionsSince: 7, - lastSessionId: "session-abc", - }; - setDreamState(persisted); - - const state = getDreamState(STATE_DIR); - expect(state).toEqual(persisted); - }); -}); diff --git a/integrations/openclaw/tsconfig.json b/integrations/openclaw/tsconfig.json index 2292f3295..1968e88d2 100644 --- a/integrations/openclaw/tsconfig.json +++ b/integrations/openclaw/tsconfig.json @@ -7,7 +7,7 @@ "declarationMap": true, "sourceMap": true, "outDir": "dist", - "rootDir": ".", + "rootDir": "..", "strict": false, "noImplicitAny": false, "types": ["node"], @@ -19,6 +19,6 @@ "allowImportingTsExtensions": true, "noEmit": true }, - "include": ["index.ts", "types.ts", "providers.ts", "config.ts", "filtering.ts", "isolation.ts", "openclaw-plugin-sdk.d.ts", "backend/**/*.ts", "tools/**/*.ts", "cli/**/*.ts", "skill-loader.ts", "recall.ts", "dream-gate.ts", "telemetry.ts", "fs-safe.ts"], + "include": ["index.ts", "types.ts", "providers.ts", "config.ts", "filtering.ts", "isolation.ts", "openclaw-plugin-sdk.d.ts", "backend/**/*.ts", "tools/**/*.ts", "cli/**/*.ts", "skill-loader.ts", "recall.ts", "telemetry.ts", "fs-safe.ts"], "exclude": ["node_modules", "dist", "**/*.test.ts"] } diff --git a/integrations/openclaw/types.ts b/integrations/openclaw/types.ts index 29d552396..caebf3fef 100644 --- a/integrations/openclaw/types.ts +++ b/integrations/openclaw/types.ts @@ -87,17 +87,6 @@ export interface SkillsConfig { identityAlwaysInclude?: boolean; categoryOrder?: string[]; }; - dream?: { - enabled?: boolean; - /** Enable automatic triggering based on activity gates. Default: true when dream enabled. */ - auto?: boolean; - /** Minimum hours between consolidations. Default: 24. */ - minHours?: number; - /** Minimum interactive sessions before triggering. Default: 5. */ - minSessions?: number; - /** Minimum total memories to justify consolidation. Default: 20. */ - minMemories?: number; - }; domain?: string; customRules?: { include?: string[]; diff --git a/integrations/mem0-plugin/.opencode-plugin/LICENSE b/integrations/opencode-plugin/LICENSE similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/LICENSE rename to integrations/opencode-plugin/LICENSE diff --git a/integrations/mem0-plugin/.opencode-plugin/README.md b/integrations/opencode-plugin/README.md similarity index 88% rename from integrations/mem0-plugin/.opencode-plugin/README.md rename to integrations/opencode-plugin/README.md index b4ba842a7..7199af3f3 100644 --- a/integrations/mem0-plugin/.opencode-plugin/README.md +++ b/integrations/opencode-plugin/README.md @@ -13,7 +13,7 @@ This adds the plugin to your `~/.config/opencode/opencode.json`. The plugin regi **Or let your agent do it** — paste this into OpenCode: ``` -Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/integrations/mem0-plugin/.opencode-plugin/README.md +Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/integrations/opencode-plugin/README.md ``` Get your API key (free): [app.mem0.ai/dashboard/api-keys](https://app.mem0.ai/dashboard/api-keys) @@ -30,7 +30,7 @@ Restart OpenCode. |-----------|-------------| | **9 Native Memory Tools** | `add_memory`, `search_memories`, `get_memories`, `update_memory`, `delete_memory`, and more — registered as OpenCode tools, backed by the `mem0ai` SDK (no MCP server required) | | **Lifecycle Hooks** | Auto-search on session start and every prompt, error memory lookup, compaction context, secret redaction | -| **9 Skills** | `/mem0-remember`, `/mem0-tour`, `/mem0-search`, `/mem0-status`, `/mem0-scope`, `/mem0-dream`, `/mem0-forget`, `/mem0-pin`, `/mem0-context-loader` — discovered in place from the plugin via OpenCode's `skills.paths` | +| **7 Skills** | `/mem0-remember`, `/mem0-tour`, `/mem0-search`, `/mem0-status`, `/mem0-scope`, `/mem0-forget`, `/mem0-context-loader` — discovered in place from the plugin via OpenCode's `skills.paths` | ## Hooks @@ -59,6 +59,7 @@ Pure TypeScript — no Python, no shell scripts. Memory operations are native Op | `delete_all_memories` | Bulk delete all memories in scope | | `delete_entities` | Delete an entity and its memories | | `list_entities` | List users/agents/apps stored in Mem0 | +| `get_event_status` | Check the processing status of an asynchronous memory event | ## Memory scope @@ -69,7 +70,7 @@ scope (used when none is passed) with the `/mem0-scope` skill: |-------|-------|--------| | `project` (default) | this repo (`user_id` + `app_id`) | this repo | | `session` | this run (adds `run_id`) | this run | -| `global` | all your projects (`app_id="*"`) | user-wide (drops `app_id`) | +| `global` | all your projects (filtered by your user ID) | user-wide (drops `app_id`) | ``` /mem0-scope # show the current default scope @@ -99,3 +100,5 @@ If the `mem0` tools respond, you're all set. ## License Apache-2.0 + +A memory tool cannot select global scope unless `/mem0-scope global` or the plugin settings already enable it. diff --git a/integrations/mem0-plugin/.opencode-plugin/api-key.test.ts b/integrations/opencode-plugin/api-key.test.ts similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/api-key.test.ts rename to integrations/opencode-plugin/api-key.test.ts diff --git a/integrations/mem0-plugin/.opencode-plugin/api-key.ts b/integrations/opencode-plugin/api-key.ts similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/api-key.ts rename to integrations/opencode-plugin/api-key.ts diff --git a/integrations/mem0-plugin/.opencode-plugin/bun.lock b/integrations/opencode-plugin/bun.lock similarity index 94% rename from integrations/mem0-plugin/.opencode-plugin/bun.lock rename to integrations/opencode-plugin/bun.lock index 9da6adb40..744a408ca 100644 --- a/integrations/mem0-plugin/.opencode-plugin/bun.lock +++ b/integrations/opencode-plugin/bun.lock @@ -12,9 +12,6 @@ "bun-types": ">=1.3.14", "typescript": "^5.7.3", }, - "peerDependencies": { - "bun": ">=1.0.0", - }, }, }, "packages": { @@ -88,38 +85,6 @@ "@opencode-ai/sdk": ["@opencode-ai/sdk@1.15.11", "", { "dependencies": { "cross-spawn": "7.0.6" } }, "sha512-IyYyDVsO8SKbKbkSadHpDuYnYC+2vmEeLU+rW+rH2M54Sigq6l3gDHno16+U6SRut+lowbph7v/ry3WbV67V3w=="], - "@oven/bun-darwin-aarch64": ["@oven/bun-darwin-aarch64@1.3.14", "", { "os": "darwin", "cpu": "arm64" }, "sha512-Omj20SuiHBOUjUBIyqtkNjSUIjOtEOJwmbix/ZyFH4BaQ6OZTaaRWIR4TjHVz0yadHgli6lLTiAh1uarnvD49A=="], - - "@oven/bun-darwin-x64": ["@oven/bun-darwin-x64@1.3.14", "", { "os": "darwin", "cpu": "x64" }, "sha512-FFj3QdU/OhlDyZOJ8CWfN5eWLpRlT4qjZg7lMQi7jA6GuoY5ajlO1zWLP/MuHYRSbXQUvV52RejNi8DVnAp13w=="], - - "@oven/bun-darwin-x64-baseline": ["@oven/bun-darwin-x64-baseline@1.3.14", "", { "os": "darwin", "cpu": "x64" }, "sha512-OSfsTZstc898HHElhU4NccaBGOSSDn5VfahiVTnidZ9B/+wb7WTyfZJaBeJcfjwJ9H2W9uTh2TGtl3UfcXgV9g=="], - - "@oven/bun-freebsd-aarch64": ["@oven/bun-freebsd-aarch64@1.3.14", "", { "os": "freebsd", "cpu": "arm64" }, "sha512-LIKrXaFxAHybVO5Pf+9XP2FHUj/5APvXTUKk9dqHm5iFz4oH+W24cmhjkJirNujh9hKeTyrpWSe3no9JZKowIw=="], - - "@oven/bun-freebsd-x64": ["@oven/bun-freebsd-x64@1.3.14", "", { "os": "freebsd", "cpu": "x64" }, "sha512-uwD+fGUH1ADpIF3B1U2jWzzb20QwRLZfj5QZ28GUCGrAJ/nTmWrD6YYGsblCY1wuhldRez3lU40AyuvSCyLYmw=="], - - "@oven/bun-linux-aarch64": ["@oven/bun-linux-aarch64@1.3.14", "", { "os": "linux", "cpu": "arm64" }, "sha512-X5SsPZHs+iYO8R/efIcRtc7gT2Q2DgPfliCxEkx4cXBumwkw0c/EsHMNwH3EgGpCDaZ7IYVPhpCG/xBOQHEwZw=="], - - "@oven/bun-linux-aarch64-android": ["@oven/bun-linux-aarch64-android@1.3.14", "", { "os": "android", "cpu": "arm64" }, "sha512-y4kq5b85lsrmFb9Xvi4w9mA5IEFJkLMrSmYn06q24KjL9rUWDWO3VFZEtteZxUN5+ec3Zm5S8OnJw1umaCbVjA=="], - - "@oven/bun-linux-aarch64-musl": ["@oven/bun-linux-aarch64-musl@1.3.14", "", { "os": "linux", "cpu": "arm64" }, "sha512-jmqOA92Cd1NL/1XBd4bFkJLxQ86K0RW7ohxS2qzzAvuitO4JiIxjjTeCspoU44zCozH72HpfZfUE2On31OjnWA=="], - - "@oven/bun-linux-x64": ["@oven/bun-linux-x64@1.3.14", "", { "os": "linux", "cpu": "x64" }, "sha512-7OVTAKvwfPmSbIV1HpdOoVVx5VRc427GuPPne93N6vk4eQBPId9nXmZDh9/zGaKPdbVjVtQSZafWQoUjx38Utw=="], - - "@oven/bun-linux-x64-android": ["@oven/bun-linux-x64-android@1.3.14", "", { "os": "android", "cpu": "x64" }, "sha512-qe9e1d+3VAEU7nAA2ol9Jvmy/o99PVMSgZhHn7Q/9O3YcDrfEqyQ8zm4zoe5qTEo8HZH0dN03Le0Ys2eQPs7eg=="], - - "@oven/bun-linux-x64-baseline": ["@oven/bun-linux-x64-baseline@1.3.14", "", { "os": "linux", "cpu": "x64" }, "sha512-q/8EdOC0yUE8FPeoOVq8/Pw5I9/tJaYmUfO/uDUAREx8IUnOJH1RJ5A3BjFqre8pvJoiZA9AovPJq5FnNNjSxA=="], - - "@oven/bun-linux-x64-musl": ["@oven/bun-linux-x64-musl@1.3.14", "", { "os": "linux", "cpu": "x64" }, "sha512-GBCB/k/sIqcr06eTNgg7g46qiUv35Jasx4XiccJ/n7RGqrE4RWUD/XJBbWFprVPjvqd59+QtSnS99XGqvftHfg=="], - - "@oven/bun-linux-x64-musl-baseline": ["@oven/bun-linux-x64-musl-baseline@1.3.14", "", { "os": "linux", "cpu": "x64" }, "sha512-n6iE71G4lQE4XkrZhQQcL5YUlxDbnq6nqV7zeQi33PMsLT/0kYE+RvHOtBWZ3w0wMdXZfINmp63hIb9ijUBGtw=="], - - "@oven/bun-windows-aarch64": ["@oven/bun-windows-aarch64@1.3.14", "", { "os": "win32", "cpu": "arm64" }, "sha512-T7s3x/BsVKQObGU6QDkZeI6wKynzqGbBH1yI77jrrj5siElclxr3DQrDIk8CV4G5/SJq2HHq4kpLyYY2DKCSmA=="], - - "@oven/bun-windows-x64": ["@oven/bun-windows-x64@1.3.14", "", { "os": "win32", "cpu": "x64" }, "sha512-mUFWL3BoYkNpjd8e9PqROiFF/1Xeotq20mABJsiQH62jM1g5zqWh4khw1RZ6bX8Q8fWvlPaxG1PjofkmjUi3vg=="], - - "@oven/bun-windows-x64-baseline": ["@oven/bun-windows-x64-baseline@1.3.14", "", { "os": "win32", "cpu": "x64" }, "sha512-uIjLUC1S9DWgICzuoMba7vurBJnBruE4S5CxnvmZkdqWVXRzx1Rgu636HoH+k0qeaQCFh3jeG3JQ1y6fRHv0sw=="], - "@protobufjs/aspromise": ["@protobufjs/aspromise@1.1.2", "", {}, "sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ=="], "@protobufjs/base64": ["@protobufjs/base64@1.1.2", "", {}, "sha512-AZkcAA5vnN/v4PDqKyMR5lx7hZttPDgClv83E//FMNhR2TMcLUhfRUBHCmSl0oi9zMgDDqRUJkSxO3wm85+XLg=="], @@ -244,8 +209,6 @@ "buffer-writer": ["buffer-writer@2.0.0", "", {}, "sha512-a7ZpuTZU1TRtnwyCNW3I5dc0wWNC3VR9S++Ewyk2HHZdrO3CQJqSpd+95Us590V6AL7JqUAH2IwZ/398PmNFgw=="], - "bun": ["bun@1.3.14", "", { "optionalDependencies": { "@oven/bun-darwin-aarch64": "1.3.14", "@oven/bun-darwin-x64": "1.3.14", "@oven/bun-darwin-x64-baseline": "1.3.14", "@oven/bun-freebsd-aarch64": "1.3.14", "@oven/bun-freebsd-x64": "1.3.14", "@oven/bun-linux-aarch64": "1.3.14", "@oven/bun-linux-aarch64-android": "1.3.14", "@oven/bun-linux-aarch64-musl": "1.3.14", "@oven/bun-linux-x64": "1.3.14", "@oven/bun-linux-x64-android": "1.3.14", "@oven/bun-linux-x64-baseline": "1.3.14", "@oven/bun-linux-x64-musl": "1.3.14", "@oven/bun-linux-x64-musl-baseline": "1.3.14", "@oven/bun-windows-aarch64": "1.3.14", "@oven/bun-windows-x64": "1.3.14", "@oven/bun-windows-x64-baseline": "1.3.14" }, "os": [ "!aix", "!sunos", "!openbsd", ], "cpu": [ "x64", "arm64", ], "bin": { "bun": "bin/bun.exe", "bunx": "bin/bunx.exe" } }, "sha512-aB6GVd42x1Y5ie1K16SF+oLGtgSkwX9hgoDdIW88pjvfTccU8F1vfpoOt34QLv0dZ1v3XimtaxPlZUG81Gx9Zg=="], - "bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="], "bundle-name": ["bundle-name@4.1.0", "", { "dependencies": { "run-applescript": "^7.0.0" } }, "sha512-tjwM5exMg6BGRI+kNmTntNsvdZS1X8BFYS6tnJ2hdH0kVxM6/eVZ2xy+FqStSWvYmtfFMDLIxurorHwDKfDz5Q=="], diff --git a/integrations/opencode-plugin/index.d.ts b/integrations/opencode-plugin/index.d.ts new file mode 100644 index 000000000..607d13c2d --- /dev/null +++ b/integrations/opencode-plugin/index.d.ts @@ -0,0 +1,4 @@ +import type { Plugin } from "@opencode-ai/plugin"; + +declare const Mem0Plugin: Plugin; +export default Mem0Plugin; diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-mem0.ts b/integrations/opencode-plugin/opencode-mem0.ts similarity index 92% rename from integrations/mem0-plugin/.opencode-plugin/opencode-mem0.ts rename to integrations/opencode-plugin/opencode-mem0.ts index d5f2acb30..8dcf6d535 100644 --- a/integrations/mem0-plugin/.opencode-plugin/opencode-mem0.ts +++ b/integrations/opencode-plugin/opencode-mem0.ts @@ -1,3 +1,4 @@ +import { resolveToolScope } from "../agent-plugin-core/typescript/src/scoping.ts"; // Mem0 memory plugin for OpenCode: captures and recalls memories across sessions // (add / search / manage) via the Mem0 platform, wired through OpenCode plugin hooks. // Memory operations are exposed as native OpenCode tools backed by the mem0ai SDK @@ -13,19 +14,10 @@ import {homedir} from "os"; import {join} from "path"; import {createHash} from "crypto"; import {captureEvent} from "./telemetry"; -import { - loadDreamConfig, - incrementSessionCount, - checkCheapGates, - checkMemoryGate, - acquireDreamLock, - releaseDreamLock, - recordDreamCompletion, - DREAM_PROTOCOL, -} from "./dream"; import {asScope, scopeSearchFilters, scopeWriteParams, resolveDefaultScope, SCOPE_GUIDANCE, type Scope} from "./scope"; import {parseProjectFromRemote} from "./project"; import {resolveApiKey} from "./api-key"; +import {createMemoryLifecycle} from "../agent-plugin-core/typescript/src/lifecycle.ts"; async function getUserId(): Promise { if (process.env.MEM0_USER_ID) return process.env.MEM0_USER_ID; @@ -78,23 +70,6 @@ function generateSessionId(): string { return `ses_${ts}_${rnd}`; } -const SECRET_PATTERNS = [ - /sk-[A-Za-z0-9]{20,}/g, - /m0-[A-Za-z0-9]{20,}/g, - /AKIA[0-9A-Z]{16}/g, - /xox[baprs]-[A-Za-z0-9-]{20,}/g, - /ghp_[A-Za-z0-9]{36,}/g, - /gho_[A-Za-z0-9]{36,}/g, -]; - -function redact(text: string): string { - let out = text; - for (const re of SECRET_PATTERNS) { - out = out.replace(re, "[REDACTED]"); - } - return out; -} - /** Read & parse `~/.mem0/settings.json`, returning {} when missing/invalid. */ function loadSettings(): Record { try { @@ -259,7 +234,7 @@ function extractUserText(input: any, output: any): string { const Mem0Plugin: Plugin = async (ctx) => { const {$, client} = ctx; - const apiKey = resolveApiKey(); + const apiKey = resolveApiKey(process.env, process.env.HOME || process.env.USERPROFILE || homedir()); if (!apiKey) { try { @@ -283,6 +258,8 @@ const Mem0Plugin: Plugin = async (ctx) => { const stats = {adds: 0, searches: 0, messages: 0}; const sessionId = generateSessionId(); const globalSearch = loadGlobalSearch(); + const lifecycle = createMemoryLifecycle(); + lifecycle.beginSession(); let initialized = false; let memoryCount = 0; @@ -290,12 +267,6 @@ const Mem0Plugin: Plugin = async (ctx) => { const systemContext: string[] = []; - // Auto-dream: gated memory-consolidation state (ported from the pi-agent plugin). - const mem0StateDir = join(homedir(), ".mem0"); - const dreamConfig = loadDreamConfig(mem0StateDir); - let dreamTriggered = false; - let dreamWriteSeen = false; - // Emit a session_stop telemetry event once when the process winds down. let sessionStopSent = false; const emitSessionStop = () => { @@ -307,16 +278,6 @@ const Mem0Plugin: Plugin = async (ctx) => { apiKey, appId, ); - // Finish an in-flight auto-dream: record completion if the agent consolidated, - // and always release the lock so the next eligible session can dream. - if (dreamTriggered) { - if (dreamWriteSeen) { - recordDreamCompletion(mem0StateDir); - captureEvent("dream_completed", {}, apiKey, appId); - } - releaseDreamLock(mem0StateDir); - dreamTriggered = false; - } }; try { process.on("beforeExit", emitSessionStop); @@ -366,7 +327,7 @@ Identity context (resolved at plugin startup): // user's persisted default scope (read fresh so /mem0-scope applies at once). // A "project" default preserves the existing behavior, including global_search. function readScopeFilters(args: any): any { - if (args.scope) return scopeSearchFilters(asScope(args.scope), userId, appId, sessionId); + if (args.scope) return scopeSearchFilters(resolveToolScope(asScope(args.scope), loadDefaultScope()), userId, appId, sessionId); if (args.filters || args.agent_id) return resolveFilters(args, globalSearch, userId, appId); const ds = loadDefaultScope(); return ds === "project" @@ -428,9 +389,8 @@ Identity context (resolved at plugin startup): }, async execute(args) { stats.adds++; - if (dreamTriggered) dreamWriteSeen = true; captureEvent("tool_use", {tool: "add_memory"}, apiKey, appId); - const effScope: Scope = args.scope ? asScope(args.scope) : loadDefaultScope(); + const effScope: Scope = resolveToolScope(args.scope ? asScope(args.scope) : undefined, loadDefaultScope()); const sp = scopeWriteParams(effScope, userId, appId, sessionId); const finalUserId = args.agent_id ? args.user_id : (args.user_id ?? sp.user_id); const finalAppId = args.app_id ?? sp.app_id; @@ -449,7 +409,7 @@ Identity context (resolved at plugin startup): } const res = await mem0.add( - [{ role: "user", content: args.text }], + [{ role: "user", content: lifecycle.prepareUserText(args.text) }], { user_id: finalUserId, app_id: finalAppId, @@ -481,7 +441,7 @@ Identity context (resolved at plugin startup): const topK = args.limit ?? args.top_k ?? 10; const filters = readScopeFilters(args); - const res = await mem0.search(args.query, { + const res = await mem0.search(lifecycle.prepareUserText(args.query), { filters, topK, }); @@ -535,7 +495,10 @@ Identity context (resolved at plugin startup): async execute(args) { captureEvent("tool_use", {tool: "update_memory"}, apiKey, appId); const res = await mem0.update(args.id, { - text: args.text, + text: + args.text === undefined + ? undefined + : lifecycle.prepareUserText(args.text), metadata: args.metadata, }); return JSON.stringify(res); @@ -548,7 +511,6 @@ Identity context (resolved at plugin startup): id: tool.schema.string().describe("The ID of the memory to delete"), }, async execute(args) { - if (dreamTriggered) dreamWriteSeen = true; captureEvent("tool_use", {tool: "delete_memory"}, apiKey, appId); const res = await mem0.delete(args.id); return JSON.stringify(res); @@ -564,9 +526,8 @@ Identity context (resolved at plugin startup): scope: tool.schema.string().optional().describe('Scope to delete: "project" (default), "session", or "global" (user-wide). Use "global" only when explicitly asked.'), }, async execute(args) { - if (dreamTriggered) dreamWriteSeen = true; captureEvent("tool_use", {tool: "delete_all_memories"}, apiKey, appId); - const sp = args.scope ? scopeWriteParams(asScope(args.scope), userId, appId, sessionId) : null; + const sp = args.scope ? scopeWriteParams(resolveToolScope(asScope(args.scope), loadDefaultScope()), userId, appId, sessionId) : null; const res = await mem0.deleteAll({ user_id: sp ? sp.user_id : (args.agent_id ? args.user_id : (args.user_id ?? userId)), app_id: sp ? sp.app_id : (args.app_id ?? appId), @@ -631,17 +592,13 @@ Identity context (resolved at plugin startup): const userText = extractUserText(input, output); if (!userText || userText.length < 10) return; - const safeText = redact(userText); + const safeText = lifecycle.prepareUserText(userText); msgCount++; stats.messages++; if (!initialized) { initialized = true; - if (dreamConfig.enabled) { - incrementSessionCount(mem0StateDir, sessionId); - } - const searchFilters = globalSearch ? {OR: [{user_id: "*"}]} : {AND: [{user_id: userId}, {app_id: appId}]}; @@ -727,28 +684,6 @@ Identity context (resolved at plugin startup): captureEvent("session_start", {memory_count: memoryCount}, apiKey, appId); - // Auto-dream: when the time/session/memory gates pass, inject the - // consolidation protocol so the agent tidies memories before answering. - if (dreamConfig.enabled && dreamConfig.auto && !dreamTriggered) { - const gates = checkCheapGates(mem0StateDir, dreamConfig); - const memGate = checkMemoryGate(memoryCount, dreamConfig); - if (gates.proceed && memGate.pass && acquireDreamLock(mem0StateDir)) { - dreamTriggered = true; - systemContext.push(DREAM_PROTOCOL); - captureEvent("dream_triggered", {memory_count: memoryCount}, apiKey, appId); - } else { - // Make "why didn't auto-dream run?" answerable from the logs. - const waiting = [gates.reason, memGate.reason].filter(Boolean).join("; "); - if (waiting) { - try { - await client.app.log({ - body: {service: "mem0", level: "info", message: `auto-dream waiting — ${waiting}`}, - }); - } catch { - } - } - } - } } const hasRemember = NUDGE_RE.test(safeText); diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-context-loader/SKILL.md b/integrations/opencode-plugin/opencode-skills/mem0-context-loader/SKILL.md similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-context-loader/SKILL.md rename to integrations/opencode-plugin/opencode-skills/mem0-context-loader/SKILL.md diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-forget/SKILL.md b/integrations/opencode-plugin/opencode-skills/mem0-forget/SKILL.md similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-forget/SKILL.md rename to integrations/opencode-plugin/opencode-skills/mem0-forget/SKILL.md diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-remember/SKILL.md b/integrations/opencode-plugin/opencode-skills/mem0-remember/SKILL.md similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-remember/SKILL.md rename to integrations/opencode-plugin/opencode-skills/mem0-remember/SKILL.md diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-scope/SKILL.md b/integrations/opencode-plugin/opencode-skills/mem0-scope/SKILL.md similarity index 98% rename from integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-scope/SKILL.md rename to integrations/opencode-plugin/opencode-skills/mem0-scope/SKILL.md index 776f4b364..1e2128141 100644 --- a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-scope/SKILL.md +++ b/integrations/opencode-plugin/opencode-skills/mem0-scope/SKILL.md @@ -75,7 +75,7 @@ Look at the user's message for a target scope word: `project`, `session`, or 3. Write the file back with the Write tool, keeping ALL existing keys and only setting `"default_scope"` to the target. Pretty-print with 2-space indent and - a trailing newline. Do not drop `global_search`, `dream`, `auto_save`, or any + a trailing newline. Do not drop `global_search`, `auto_save`, or any other field that was present. Example resulting file (when other keys already existed): diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-search/SKILL.md b/integrations/opencode-plugin/opencode-skills/mem0-search/SKILL.md similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-search/SKILL.md rename to integrations/opencode-plugin/opencode-skills/mem0-search/SKILL.md diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-status/SKILL.md b/integrations/opencode-plugin/opencode-skills/mem0-status/SKILL.md similarity index 69% rename from integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-status/SKILL.md rename to integrations/opencode-plugin/opencode-skills/mem0-status/SKILL.md index 89a0e4542..9c1d6f151 100644 --- a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-status/SKILL.md +++ b/integrations/opencode-plugin/opencode-skills/mem0-status/SKILL.md @@ -79,36 +79,6 @@ echo "branch=${MEM0_BRANCH:-}" - If all three are non-empty: PASS — "Session active" - If any are missing: WARN — "Plugin env vars not set; shell.env hook may not have fired" -### Check 6: Auto-dream readiness - -Explain whether auto-dream (memory consolidation) is eligible to run, and if not, exactly which gate is blocking. Auto-dream runs at most once per session and only when **all** gates pass: time since last consolidation ≥ `minHours`, sessions since ≥ `minSessions`, and project memory count ≥ `minMemories`. - -Read the gate state and thresholds: - -```bash -_ST="$HOME/.mem0/mem0-dream-state.json" -_SET="$HOME/.mem0/settings.json" -echo "sessions_since=$(grep -o '"sessionsSince"[[:space:]]*:[[:space:]]*[0-9]*' "$_ST" 2>/dev/null | grep -o '[0-9]*$' || echo 0)" -echo "last_consolidated_ms=$(grep -o '"lastConsolidatedAt"[[:space:]]*:[[:space:]]*[0-9]*' "$_ST" 2>/dev/null | grep -o '[0-9]*$' || echo 0)" -echo "min_hours=$(grep -o '"minHours"[[:space:]]*:[[:space:]]*[0-9]*' "$_SET" 2>/dev/null | grep -o '[0-9]*$' || echo 24)" -echo "min_sessions=$(grep -o '"minSessions"[[:space:]]*:[[:space:]]*[0-9]*' "$_SET" 2>/dev/null | grep -o '[0-9]*$' || echo 5)" -echo "min_memories=$(grep -o '"minMemories"[[:space:]]*:[[:space:]]*[0-9]*' "$_SET" 2>/dev/null | grep -o '[0-9]*$' || echo 20)" -echo "now_s=$(date +%s)" -echo "dream_env=${MEM0_DREAM:-unset}" -``` - -For the memory count, reuse the project memory count from Check 3/4 (or call `get_memories` with the project filter, `page_size=1`, and read `count`). - -Compute each gate: -- **time**: `hours_since = (now_s - last_consolidated_ms/1000) / 3600`. Passes when `≥ min_hours`. If `last_consolidated_ms` is 0 it has never run → time gate passes. -- **sessions**: passes when `sessions_since ≥ min_sessions`. -- **memories**: passes when project memory count `≥ min_memories`. - -Report: -- If `dream_env` is `false`/`0`/`no`/`off`, or `dream.enabled` is false in settings: WARN — "Auto-dream disabled". -- If all three gates pass: PASS — "eligible (runs at next session start)". -- Otherwise: WARN — list the blocking gate(s), e.g. `sessions 2/5, memories 3/20`. This is expected, not an error — auto-dream is just waiting. Note the user can run `/mem0-dream` to consolidate now, or lower the thresholds via the `dream` block in `~/.mem0/settings.json`. - ### Display ``` @@ -120,18 +90,15 @@ PASS Default scope project PASS Memory Tools 142ms PASS Write/Read write + delete OK PASS Session session_id=abc123, app_id=mem0, branch=main -WARN Auto-dream waiting — sessions 2/5, memories 3/20 (/mem0-dream to run now) All checks passed. ``` -The Auto-dream line is informational: WARN here means "waiting on gates", not a failure. Show PASS when eligible, or "disabled" when turned off. - If any check fails, add a `## Troubleshooting` section with specific fix steps for each failure. ## Extended mode: Memory Quality Analysis -When invoked with `--deep` (e.g., `/mem0-status --deep`), run the standard 6 checks above **plus** a memory quality scan. +When invoked with `--deep` (e.g., `/mem0-status --deep`), run the standard 5 checks above **plus** a memory quality scan. ### Quality Check 1: Duplicates @@ -188,9 +155,7 @@ Duplicates: · Stale: · Contradictions: · Orphans: ``` If all counts are 0: `Memory quality: clean.` -If any non-zero: append `Run /mem0-dream to fix.` - -To fix issues found by `--deep`, run `/mem0-dream` for automated consolidation (merges, prunes, conflict resolution). +If any non-zero, list the affected memory IDs so the user can review them with the normal search, update, and delete tools. ## Output formatting diff --git a/integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-tour/SKILL.md b/integrations/opencode-plugin/opencode-skills/mem0-tour/SKILL.md similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/opencode-skills/mem0-tour/SKILL.md rename to integrations/opencode-plugin/opencode-skills/mem0-tour/SKILL.md diff --git a/integrations/mem0-plugin/.opencode-plugin/package.json b/integrations/opencode-plugin/package.json similarity index 81% rename from integrations/mem0-plugin/.opencode-plugin/package.json rename to integrations/opencode-plugin/package.json index 2bd5fefd2..79f68c4a8 100644 --- a/integrations/mem0-plugin/.opencode-plugin/package.json +++ b/integrations/opencode-plugin/package.json @@ -1,13 +1,13 @@ { "name": "@mem0/opencode-plugin", - "version": "0.2.2", + "version": "0.3.0", "type": "module", "description": "Mem0 persistent memory plugin for OpenCode — add, search, and manage memories across sessions", "main": "dist/index.js", - "types": "dist/index.d.ts", + "types": "index.d.ts", "exports": { ".": { - "types": "./dist/index.d.ts", + "types": "./index.d.ts", "import": "./dist/index.js" } }, @@ -26,15 +26,16 @@ "repository": { "type": "git", "url": "https://github.com/mem0ai/mem0", - "directory": "integrations/mem0-plugin/.opencode-plugin" + "directory": "integrations/opencode-plugin" }, "files": [ "dist", + "index.d.ts", "LICENSE", "opencode-skills" ], "scripts": { - "build": "bun build opencode-mem0.ts --outdir dist --target bun --format esm --entry-naming index.[ext] && tsc --emitDeclarationOnly --outDir dist --declaration", + "build": "bun build opencode-mem0.ts --outdir dist --target bun --format esm --entry-naming index.[ext]", "dev": "bun build opencode-mem0.ts --outdir dist --target bun --format esm --entry-naming index.[ext] --watch", "type-check": "tsc --noEmit", "prepack": "bun run build", @@ -59,8 +60,5 @@ "devDependencies": { "bun-types": ">=1.3.14", "typescript": "^5.7.3" - }, - "peerDependencies": { - "bun": ">=1.0.0" } } diff --git a/integrations/mem0-plugin/.opencode-plugin/project.test.ts b/integrations/opencode-plugin/project.test.ts similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/project.test.ts rename to integrations/opencode-plugin/project.test.ts diff --git a/integrations/opencode-plugin/project.ts b/integrations/opencode-plugin/project.ts new file mode 100644 index 000000000..4e5be585d --- /dev/null +++ b/integrations/opencode-plugin/project.ts @@ -0,0 +1 @@ +export { parseProjectFromRemote } from "../agent-plugin-core/typescript/src/identity.ts"; diff --git a/integrations/mem0-plugin/.opencode-plugin/scope.test.ts b/integrations/opencode-plugin/scope.test.ts similarity index 99% rename from integrations/mem0-plugin/.opencode-plugin/scope.test.ts rename to integrations/opencode-plugin/scope.test.ts index 49718b469..32b45a9cd 100644 --- a/integrations/mem0-plugin/.opencode-plugin/scope.test.ts +++ b/integrations/opencode-plugin/scope.test.ts @@ -29,7 +29,6 @@ describe("memory scope (pi-agent parity)", () => { test("global scope spans all the user's projects (matches pi-agent)", () => { expect(scopeSearchFilters("global", "u", "app", "run")).toEqual({ user_id: "u", - app_id: "*", }); // global writes drop app_id so the memory is user-wide, not project-bound expect(scopeWriteParams("global", "u", "app", "run")).toEqual({ user_id: "u" }); diff --git a/integrations/opencode-plugin/scope.ts b/integrations/opencode-plugin/scope.ts new file mode 100644 index 000000000..eaa4521c7 --- /dev/null +++ b/integrations/opencode-plugin/scope.ts @@ -0,0 +1,42 @@ +import { + normalizeScope, + scopeAddParams, + scopeSearchFilters as sharedSearchFilters, + type Scope, +} from "../agent-plugin-core/typescript/src/scoping.ts"; + +export type { Scope }; + +const context = (userId: string, appId: string, runId: string) => ({ userId, appId, runId }); + +export function scopeSearchFilters( + scope: Scope, + userId: string, + appId: string, + runId: string, +): Record { + return sharedSearchFilters(scope, context(userId, appId, runId)); +} + +export function scopeWriteParams( + scope: Scope, + userId: string, + appId: string, + runId: string, +): { user_id: string; app_id?: string; run_id?: string } { + const values = scopeAddParams(scope, context(userId, appId, runId)); + return { + user_id: values.userId, + ...(values.appId ? { app_id: values.appId } : {}), + ...(values.runId ? { run_id: values.runId } : {}), + }; +} + +export const asScope = normalizeScope; + +export function resolveDefaultScope(settings: Record | null | undefined): Scope { + return normalizeScope(settings?.default_scope); +} + +export const SCOPE_GUIDANCE = + 'Memory tools accept an optional `scope`: omit it (or "project") for normal queries; use "session" to limit to the current run; use "global" only after the user enables /mem0-scope global.'; diff --git a/integrations/mem0-plugin/.opencode-plugin/telemetry.test.ts b/integrations/opencode-plugin/telemetry.test.ts similarity index 100% rename from integrations/mem0-plugin/.opencode-plugin/telemetry.test.ts rename to integrations/opencode-plugin/telemetry.test.ts diff --git a/integrations/opencode-plugin/telemetry.ts b/integrations/opencode-plugin/telemetry.ts new file mode 100644 index 000000000..8d0861d97 --- /dev/null +++ b/integrations/opencode-plugin/telemetry.ts @@ -0,0 +1,66 @@ +import { createHash } from "node:crypto"; +import { readFileSync } from "node:fs"; +import { release } from "node:os"; + +import { + createTelemetry, + isTelemetryEnabled, +} from "../agent-plugin-core/typescript/src/telemetry.ts"; + +export { isTelemetryEnabled }; + +const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"; +const PLUGIN_VERSION = (() => { + for (const relative of ["./package.json", "../package.json"]) { + try { + const pkg = JSON.parse(readFileSync(new URL(relative, import.meta.url), "utf-8")); + if (pkg?.name === "@mem0/opencode-plugin" && pkg.version) return pkg.version; + } catch { + // Try the source or bundled location. + } + } + return "unknown"; +})(); + +let currentDistinctId = ""; +const telemetry = createTelemetry({ + host: "opencode", + source: "plugin", + version: PLUGIN_VERSION, + distinctId: () => currentDistinctId, + eventName: (event) => `plugin.${event}`, + commonProperties: { + platform: "opencode", + os_version: release(), + sample_rate: 1.0, + }, +}); + +function distinctId(apiKey: string): string { + return createHash("sha256").update(apiKey).digest("hex").slice(0, 32); +} + +function projectHash(projectId?: string): Record { + return projectId ? { project_hash: createHash("sha256").update(projectId).digest("hex") } : {}; +} + +export function buildEvent( + eventType: string, + properties: Record, + apiKey: string | undefined, + projectId?: string, +): Record | null { + currentDistinctId = apiKey ? distinctId(apiKey) : ""; + const event = telemetry.build(eventType, { ...properties, ...projectHash(projectId) }); + return event ? { api_key: POSTHOG_API_KEY, ...event } : null; +} + +export function captureEvent( + eventType: string, + properties: Record, + apiKey: string | undefined, + projectId?: string, +): void { + currentDistinctId = apiKey ? distinctId(apiKey) : ""; + telemetry.capture(eventType, { ...properties, ...projectHash(projectId) }); +} diff --git a/integrations/mem0-plugin/.opencode-plugin/tsconfig.json b/integrations/opencode-plugin/tsconfig.json similarity index 88% rename from integrations/mem0-plugin/.opencode-plugin/tsconfig.json rename to integrations/opencode-plugin/tsconfig.json index a635ca045..9ff59dc82 100644 --- a/integrations/mem0-plugin/.opencode-plugin/tsconfig.json +++ b/integrations/opencode-plugin/tsconfig.json @@ -6,12 +6,13 @@ "declaration": true, "declarationDir": "dist", "outDir": "dist", - "rootDir": ".", + "rootDir": "..", "strict": true, "esModuleInterop": true, "skipLibCheck": true, "lib": ["ESNext"], "types": ["bun-types"], + "allowImportingTsExtensions": true, "noEmit": false, "emitDeclarationOnly": true }, diff --git a/integrations/pi-agent-plugin/README.md b/integrations/pi-agent-plugin/README.md index 0c88c1b6b..c4d9a8286 100644 --- a/integrations/pi-agent-plugin/README.md +++ b/integrations/pi-agent-plugin/README.md @@ -2,7 +2,7 @@ Persistent semantic memory for [Pi Agent](https://pi.dev), powered by [Mem0](https://mem0.ai). -This extension gives Pi Agent long-term memory that persists across sessions, projects, and devices. Memories are automatically captured from conversations and can be searched, managed, and consolidated through slash commands and an agent-accessible tool. +This extension gives Pi Agent long-term memory that persists across sessions, projects, and devices. Memories are automatically captured from conversations and can be searched and managed through slash commands and an agent-accessible tool. ## Features @@ -10,9 +10,8 @@ This extension gives Pi Agent long-term memory that persists across sessions, pr - **Semantic search** — find memories by meaning, not just keywords - **Scoped memory** — project, session, or global scope - **Monorepo-aware** — uses git root for project detection, consistent app_id across subdirectories -- **Dream consolidation** — merges duplicates, resolves contradictions, prunes stale entries - **Confirmation dialogs** — destructive commands ask before acting -- **8 slash commands** — essential memory management from the command line +- **6 slash commands** — essential memory management from the command line - **Agent tool** — `mem0_memory` tool lets the agent search and store memories autonomously ## Setup @@ -43,20 +42,13 @@ Or create a config file at `~/.pi/agent/mem0-config.json`: "userId": "your-username", "autoCapture": true, "defaultScope": "project", - "searchThreshold": 0.2, - "dream": { - "enabled": true, - "auto": true, - "minHours": 24, - "minSessions": 5, - "minMemories": 20 - } + "searchThreshold": 0.3 } ``` Environment variables (`MEM0_API_KEY`, `MEM0_USER_ID`) override the config file. -`searchThreshold` (default `0.3`) is the minimum similarity score (0–1) a memory must reach to count as a match for `/mem0-search`, `/mem0-forget`, and `/mem0-pin`. It is passed to the mem0 search API (along with reranking for higher-precision ordering), so a query with no sufficiently similar memory reports no match instead of returning the closest unrelated memories. Raise it to be stricter; lower it if relevant results are missed. +`searchThreshold` (default `0.3`) is the minimum similarity score (0–1) a memory must reach to count as a match for `/mem0-search` and `/mem0-forget`. It is passed to the mem0 search API (along with reranking for higher-precision ordering), so a query with no sufficiently similar memory reports no match instead of returning the closest unrelated memories. Raise it to be stricter; lower it if relevant results are missed. ## Commands @@ -66,14 +58,12 @@ Environment variables (`MEM0_API_KEY`, `MEM0_USER_ID`) override the config file. | `/mem0-forget ` | Search and delete memories (with confirmation) | | `/mem0-search ` | Semantic search across memories | | `/mem0-tour [scope]` | Browse all memories grouped by category | -| `/mem0-dream` | Consolidate — merge duplicates, prune stale, resolve contradictions | -| `/mem0-pin ` | Pin a memory to protect from dream pruning (preserves ID) | | `/mem0-scope ` | Change default scope for this session | | `/mem0-status` | Connection health, identity, and memory count | ## Skills -The plugin includes 8 skills that guide the agent on how to use each capability: +The plugin includes 6 skills that guide the agent on how to use each capability: | Skill | Purpose | |-------|---------| @@ -81,9 +71,7 @@ The plugin includes 8 skills that guide the agent on how to use each capability: | `remember` | Store facts with category classification | | `search` | Quick semantic search with compact results | | `forget` | Delete memories with confirmation | -| `dream` | Memory consolidation workflow | | `tour` | Full memory walkthrough by category | -| `pin` | Protect critical memories from pruning | | `status` | Health check and diagnostics | ## Memory Scopes @@ -120,15 +108,14 @@ pi-agent-plugin/ ├── src/ │ ├── entry.ts # Extension entry point │ ├── index.ts # Barrel exports -│ ├── commands.ts # 8 slash commands +│ ├── commands.ts # 6 slash commands │ ├── prompt.ts # System prompt injection (MEMORY_POLICY) │ ├── types.ts # Shared interfaces and categories │ ├── telemetry.ts # PostHog telemetry (batched, PII-safe) │ ├── config/ # Config loading (~/.pi/agent/mem0-config.json) │ ├── memory/ # Tool registration, scoping (git root), formatting -│ ├── capture/ # Auto-capture from conversations (user + assistant) -│ └── dream/ # Consolidation state, gating, locking, prompts -├── skills/ # 8 SKILL.md files for Pi Agent +│ └── capture/ # Auto-capture from conversations (user + assistant) +├── skills/ # 6 SKILL.md files for Pi Agent ├── tests/ # Vitest unit tests └── dist/ # Built output (ESM + DTS) ``` @@ -145,3 +132,5 @@ pnpm run build # Build (ESM + declarations) ## License [Apache-2.0](LICENSE) + +Global tool operations require `/mem0-scope global` or `defaultScope: "global"` in plugin configuration. A model-supplied `scope` argument cannot enable cross-project access on its own. Empty or wildcard user, project, and session identities are rejected. diff --git a/integrations/pi-agent-plugin/package.json b/integrations/pi-agent-plugin/package.json index 69d3d69ca..d9b3f7d2e 100644 --- a/integrations/pi-agent-plugin/package.json +++ b/integrations/pi-agent-plugin/package.json @@ -1,6 +1,6 @@ { "name": "@mem0/pi-agent-plugin", - "version": "0.1.5", + "version": "0.3.0", "type": "module", "description": "Mem0 memory extension for Pi Agent persistent, scoped, semantic memory across sessions and projects", "license": "Apache-2.0", @@ -32,14 +32,13 @@ }, "files": [ "dist", - "src", "skills", "README.md", "LICENSE" ], "pi": { "extensions": [ - "./src/entry.ts" + "./dist/entry.js" ], "skills": [ "./skills" diff --git a/integrations/pi-agent-plugin/skills/dream/SKILL.md b/integrations/pi-agent-plugin/skills/dream/SKILL.md deleted file mode 100644 index 21e9c5f43..000000000 --- a/integrations/pi-agent-plugin/skills/dream/SKILL.md +++ /dev/null @@ -1,125 +0,0 @@ ---- -name: dream -description: Consolidates stored memories by merging duplicates, resolving contradictions, and pruning stale entries. Use when memory count is high, search results feel noisy or repetitive, or periodic cleanup is needed to maintain memory quality. ---- - -# Dream — Memory Consolidation - -This skill performs a memory consolidation pass: it fetches all memories, identifies near-duplicates, flags contradictions, and prunes stale entries. All proposed changes are shown as a diff for user approval before anything is modified. - -**IMPORTANT: Execute steps strictly in order (1 -> 2 -> 3 -> 4 -> 5). Each step depends on the previous one. Do NOT run steps in parallel or skip ahead.** - -## Step 1: Fetch ALL Memories - -Use `mem0_memory` tool with `action="get_all"` to retrieve every memory. - -If zero memories are found, print: - -``` -No memories found. Nothing to consolidate. -``` - -...and stop. - -## Step 2: Analyze — Find Issues - -Work entirely in-memory; do not modify anything yet. - -Group memories by category. For each group, identify the following: - -### 2a. Near-duplicate pairs (merge candidates) - -Two memories are near-duplicates when they express the same fact but phrased differently (e.g., "Prefers morning meetings" and "Likes scheduling meetings early"). - -Heuristics — two memories are near-duplicates if **all** of these hold: -- If >60% of significant nouns/keywords overlap, treat as near-duplicate. -- Same category. -- Neither memory is pinned (content does not start with `[PINNED]`). - -For each qualifying pair, draft a merged version that is more complete than either original. - -### 2b. Contradictions - -Two memories contradict when they assert opposing facts about the same topic (e.g., "Prefers cats" vs. "Allergic to cats, prefers dogs"). - -Identify the likely winner: the more recent memory wins. Store both IDs and their content for user review. - -### 2c. Prune candidates - -A memory is a prune candidate when **any** of the following is true: - -1. It is older than 180 days AND has not been accessed recently. -2. Its content is extremely vague (fewer than 5 meaningful words). - -**Always skip memories where content starts with `[PINNED]`**, regardless of age. - -## Step 3: Print Diff Report - -Print a structured diff before making any changes: - -``` -## dream — consolidation report - -Merges (): - [mem0:] + [mem0:] -> "" - -Conflicts (): - [mem0:] vs [mem0:] — "" [A/B/skip] - -Prune (): - [mem0:] — , d old - -Proposed: merges, prunes, conflicts. Apply? [Y/n] -``` - -If there are zero total proposals, print: - -``` -Dream complete. No duplicate, contradictory, or stale memories found. -``` - -...and stop. - -## Step 4: Wait for User Input and Apply - -### 4a. Contradictions - -For each conflict pair, wait for the user to choose A, B, or skip. - -### 4b. Final confirmation - -After all conflict resolutions are collected, prompt: `Apply? [Y/n]` - -If the user declines, print `Cancelled. No changes made.` and stop. - -If confirmed, apply all changes: - -**Merges:** Delete both originals, add the merged version using `mem0_memory` with `action="add"`. - -**Contradictions (resolved):** Delete the loser using `mem0_memory` with `action="delete"`. - -**Prunes:** Delete each using `mem0_memory` with `action="delete"`. - -## Step 5: Print Summary - -``` -Dream complete — merged: , pruned: , conflicts resolved: , skipped: -``` - -## Auto mode - -When invoked with `--auto` (e.g., `/mem0-dream --auto`), run non-interactively: - -- **Merges**: applied automatically. -- **Prunes**: applied automatically. -- **Contradictions**: skipped — they require human judgment. - -Print a compact summary: -``` -[mem0-dream --auto] merged= pruned= conflicts_skipped= -``` - -## See also - -- `/mem0-forget` — targeted deletion of specific memories -- `/mem0-status` — quick health check diff --git a/integrations/pi-agent-plugin/skills/pin/SKILL.md b/integrations/pi-agent-plugin/skills/pin/SKILL.md deleted file mode 100644 index 11b075d37..000000000 --- a/integrations/pi-agent-plugin/skills/pin/SKILL.md +++ /dev/null @@ -1,48 +0,0 @@ ---- -name: pin -description: Pins or unpins a memory to protect it from pruning during dream consolidation. Use when a memory is critical and must never be removed, such as core preferences, important decisions, or immutable personal facts. ---- - -# Pin - -Pin a memory to mark it as high-priority and protect from dream pruning. - -## Execution - -### Step 1: Find the memory - -The user provides either a search query or memory ID. - -**If memory ID:** Look it up directly. - -**If search query:** -- Use `mem0_memory` tool with `action="search"`, `query=`. -- Show numbered list with content previews. -- Ask: "Which memory to pin? Enter a number." - -### Step 2: Pin it - -Pinning works by prepending `[PINNED]` to the memory text. This marker tells the dream consolidation to skip it during pruning. - -Use `mem0_memory` tool with `action="add"`, `content="[PINNED] "`. - -Then delete the original using `mem0_memory` with `action="delete"` and the original memory ID. - -**For new memories** (user wants to pin text that isn't stored yet): -- Use `mem0_memory` tool with `action="add"`, `content="[PINNED] "`. - -### Step 3: Confirm - -``` -Pinned: "" -``` - -Append `...` only if content exceeds 80 characters. - -### Unpin - -If the user says "unpin": -1. Find the memory (search or by ID). -2. Create a new memory without the `[PINNED]` prefix. -3. Delete the pinned version. -4. Print: `Unpinned: "..."` diff --git a/integrations/pi-agent-plugin/skills/status/SKILL.md b/integrations/pi-agent-plugin/skills/status/SKILL.md index 75d073fd7..5778cfe7f 100644 --- a/integrations/pi-agent-plugin/skills/status/SKILL.md +++ b/integrations/pi-agent-plugin/skills/status/SKILL.md @@ -86,4 +86,4 @@ Duplicates: · Stale: · Contradictions: ``` If all counts are 0: `Memory quality: clean.` -If any non-zero: append `Run /mem0-dream to fix.` +If any non-zero, list the affected memory IDs so the user can review them with the normal search and forget commands. diff --git a/integrations/pi-agent-plugin/src/capture/index.ts b/integrations/pi-agent-plugin/src/capture/index.ts index 3b7a5d5d4..0c35fb630 100644 --- a/integrations/pi-agent-plugin/src/capture/index.ts +++ b/integrations/pi-agent-plugin/src/capture/index.ts @@ -4,37 +4,9 @@ import type { Mem0Config, ScopeContext } from "../types.ts"; import { DEFAULT_CUSTOM_CATEGORIES } from "../types.ts"; import { resolveAddParams } from "../memory/scoping.ts"; import { captureEvent } from "../telemetry.ts"; +import { createMemoryLifecycle } from "../../../agent-plugin-core/typescript/src/lifecycle.ts"; -interface MessageLike { - role: string; - content?: unknown; -} - -function extractText(content: unknown): string | null { - if (typeof content === "string") return content; - if (Array.isArray(content)) { - const texts = content - .filter((b: any) => b.type === "text" && typeof b.text === "string") - .map((b: any) => b.text); - return texts.length > 0 ? texts.join("\n") : null; - } - return null; -} - -export function extractConversation( - messages: MessageLike[], -): Array<{ role: "user" | "assistant"; content: string }> { - const result: Array<{ role: "user" | "assistant"; content: string }> = []; - - for (const msg of messages) { - if (msg.role !== "user" && msg.role !== "assistant") continue; - const text = extractText(msg.content); - if (!text) continue; - result.push({ role: msg.role as "user" | "assistant", content: text }); - } - - return result; -} +export { extractConversation } from "../../../agent-plugin-core/typescript/src/lifecycle.ts"; export function setupAutoCapture( pi: ExtensionAPI, @@ -42,12 +14,13 @@ export function setupAutoCapture( config: Mem0Config, getScopeCtx: () => ScopeContext, telemetryCtx?: { apiKey?: string }, + lifecycle: ReturnType = createMemoryLifecycle(), ): void { if (!config.autoCapture) return; pi.on("agent_end", async (event) => { const messages = event.messages ?? []; - const conversation = extractConversation(messages); + const conversation = lifecycle.prepareConversation(messages); if (conversation.length === 0) return; const scopeCtx = getScopeCtx(); diff --git a/integrations/pi-agent-plugin/src/commands.test.ts b/integrations/pi-agent-plugin/src/commands.test.ts index 1ba46c6be..73ac9c6cb 100644 --- a/integrations/pi-agent-plugin/src/commands.test.ts +++ b/integrations/pi-agent-plugin/src/commands.test.ts @@ -6,14 +6,6 @@ vi.mock("./telemetry.ts", () => ({ captureCommandEvent: vi.fn(), })); -vi.mock("./dream/index.ts", () => ({ - acquireDreamLock: vi.fn(() => true), -})); - -vi.mock("./dream/prompt.ts", () => ({ - DREAM_PROTOCOL: "dream protocol text", -})); - function makeMem0() { return { search: vi.fn(), @@ -56,7 +48,6 @@ const defaultConfig: Mem0Config = { defaultScope: "project", contextInjection: false, searchThreshold: 0.3, - dream: { enabled: false, auto: false, minHours: 24, minSessions: 5, minMemories: 20 }, }; const scopeCtx: ScopeContext = { userId: "test-user", appId: "test-app", runId: "test-run" }; @@ -74,14 +65,14 @@ describe("registerCommands", () => { it("registers all expected commands", () => { const names = [...pi._commands.keys()]; - expect(names).toContain("mem0-remember"); - expect(names).toContain("mem0-forget"); - expect(names).toContain("mem0-search"); - expect(names).toContain("mem0-tour"); - expect(names).toContain("mem0-dream"); - expect(names).toContain("mem0-pin"); - expect(names).toContain("mem0-scope"); - expect(names).toContain("mem0-status"); + expect(names).toEqual([ + "mem0-remember", + "mem0-forget", + "mem0-search", + "mem0-tour", + "mem0-scope", + "mem0-status", + ]); }); describe("/mem0-forget", () => { @@ -197,103 +188,6 @@ describe("registerCommands", () => { }); }); - describe("/mem0-pin", () => { - it("uses update to pin in-place, preserving memory ID", async () => { - const ctx = makeCtx(true); - mem0.search.mockResolvedValue({ results: [{ id: "abc-123", memory: "important fact" }] }); - mem0.update.mockResolvedValue([]); - - await pi._invoke("mem0-pin", "important", ctx); - - expect(ctx.ui.confirm).toHaveBeenCalledWith( - "Pin this memory?", - expect.stringContaining("important fact"), - ); - expect(mem0.update).toHaveBeenCalledWith("abc-123", { text: "[PINNED] important fact" }); - expect(mem0.add).not.toHaveBeenCalled(); - expect(mem0.delete).not.toHaveBeenCalled(); - }); - - it("sends a visible confirmation after pinning", async () => { - const ctx = makeCtx(true); - mem0.search.mockResolvedValue({ results: [{ id: "abc-123", memory: "important fact" }] }); - mem0.update.mockResolvedValue([]); - - await pi._invoke("mem0-pin", "important", ctx); - - expect(pi.sendMessage).toHaveBeenCalledWith( - expect.objectContaining({ - customType: "mem0-pin", - content: expect.stringContaining("Pinned"), - display: true, - }), - ); - }); - - it("does not pin when user cancels", async () => { - const ctx = makeCtx(false); - mem0.search.mockResolvedValue({ results: [{ id: "abc-123", memory: "fact" }] }); - - await pi._invoke("mem0-pin", "fact", ctx); - - expect(mem0.update).not.toHaveBeenCalled(); - }); - - it("skips already-pinned memories with a visible message", async () => { - const ctx = makeCtx(); - mem0.search.mockResolvedValue({ results: [{ id: "abc-123", memory: "[PINNED] fact" }] }); - - await pi._invoke("mem0-pin", "fact", ctx); - - expect(ctx.ui.confirm).not.toHaveBeenCalled(); - expect(mem0.add).not.toHaveBeenCalled(); - expect(pi.sendMessage).toHaveBeenCalledWith( - expect.objectContaining({ content: expect.stringContaining("Already pinned"), display: true }), - ); - }); - - it("uses select UI for multiple matches and pins chosen memory", async () => { - const ctx = makeCtx(); - mem0.search.mockResolvedValue({ - results: [ - { id: "id-1", memory: "fact one" }, - { id: "id-2", memory: "fact two" }, - ], - }); - mem0.update.mockResolvedValue([]); - ctx.ui.select = vi.fn(async (_title: string, options: string[]) => options[1]); - - await pi._invoke("mem0-pin", "fact", ctx); - - expect(ctx.ui.select).toHaveBeenCalledWith( - expect.stringContaining("which should I pin"), - expect.arrayContaining([ - expect.stringContaining("fact one"), - expect.stringContaining("fact two"), - ]), - ); - expect(mem0.update).toHaveBeenCalledWith("id-2", { text: "[PINNED] fact two" }); - }); - - it("does not pin when user cancels select", async () => { - const ctx = makeCtx(); - ctx.ui.select = vi.fn(async () => undefined); - mem0.search.mockResolvedValue({ - results: [ - { id: "id-1", memory: "fact one" }, - { id: "id-2", memory: "fact two" }, - ], - }); - - await pi._invoke("mem0-pin", "fact", ctx); - - expect(mem0.update).not.toHaveBeenCalled(); - expect(pi.sendMessage).toHaveBeenCalledWith( - expect.objectContaining({ content: expect.stringContaining("Cancelled"), display: true }), - ); - }); - }); - describe("/mem0-search", () => { it("performs server-side semantic search with a relevance threshold", async () => { const ctx = makeCtx(); @@ -488,23 +382,4 @@ describe("registerCommands", () => { }); }); - describe("/mem0-dream", () => { - it("feeds the protocol to the agent and shows a clean status line", async () => { - const ctx = makeCtx(); - - await pi._invoke("mem0-dream", "", ctx); - - expect(pi.sendMessage).toHaveBeenCalledWith( - expect.objectContaining({ customType: "mem0-dream", display: false }), - expect.objectContaining({ triggerTurn: true }), - ); - expect(pi.sendMessage).toHaveBeenCalledWith( - expect.objectContaining({ - customType: "mem0-dream", - content: expect.stringContaining("Dreaming"), - display: true, - }), - ); - }); - }); }); diff --git a/integrations/pi-agent-plugin/src/commands.ts b/integrations/pi-agent-plugin/src/commands.ts index e5b760043..9379bed9a 100644 --- a/integrations/pi-agent-plugin/src/commands.ts +++ b/integrations/pi-agent-plugin/src/commands.ts @@ -4,9 +4,6 @@ import type { Mem0Config, ScopeContext, Scope } from "./types.ts"; import { DEFAULT_CUSTOM_CATEGORIES } from "./types.ts"; import { resolveSearchFilters, resolveAddParams } from "./memory/scoping.ts"; import { formatMemoryList, formatMemoryCompact, groupByCategory } from "./memory/formatting.ts"; -import { DREAM_PROTOCOL } from "./dream/prompt.ts"; -import { acquireDreamLock } from "./dream/index.ts"; -import { CONFIG_DIR } from "./config/index.ts"; import { captureCommandEvent } from "./telemetry.ts"; const SEARCH_TOP_K = 10; @@ -184,90 +181,6 @@ export function registerCommands( }, }); - pi.registerCommand("mem0-dream", { - description: "Consolidate memories — merge duplicates, prune stale entries, resolve contradictions", - handler: async (_args, ctx) => { - if (!acquireDreamLock(CONFIG_DIR)) { - ctx.ui.notify("A dream consolidation is already in progress.", "warning"); - return; - } - - captureCommandEvent("mem0-dream", {}, telemetryCtx); - pi.sendMessage({ customType: "mem0-dream", content: DREAM_PROTOCOL, display: false }, { triggerTurn: true }); - sendFeedback( - "mem0-dream", - "**Dreaming** — reviewing your memories to merge duplicates, resolve contradictions, and prune stale entries. I'll report what changed.", - ); - }, - }); - - pi.registerCommand("mem0-pin", { - description: "Pin a memory to protect it from dream pruning", - handler: async (args, ctx) => { - const query = args?.trim(); - if (!query) { - ctx.ui.notify("Usage: /mem0-pin ", "warning"); - return; - } - - const memories = await searchMemories(query, config.defaultScope); - - if (memories.length === 0) { - captureCommandEvent("mem0-pin", { result_count: 0 }, telemetryCtx); - sendFeedback("mem0-pin", `**No matches for "${query}"** — nothing to pin.`); - return; - } - - const pinned = (mem: Parameters[0]) => { - captureCommandEvent("mem0-pin", { pinned: true }, telemetryCtx); - sendFeedback( - "mem0-pin", - ["**Pinned** — protected from dream pruning", `- ${formatMemoryCompact(mem)}`].join("\n"), - ); - }; - const alreadyPinned = (mem: Parameters[0]) => { - sendFeedback("mem0-pin", ["**Already pinned**", `- ${formatMemoryCompact(mem)}`].join("\n")); - }; - - if (memories.length === 1) { - const target = memories[0]; - const text = target.memory ?? ""; - if (text.startsWith("[PINNED]")) { - alreadyPinned(target); - return; - } - const confirmed = await ctx.ui.confirm("Pin this memory?", formatMemoryCompact(target)); - if (!confirmed) { - sendFeedback("mem0-pin", "**Cancelled** — nothing was pinned."); - return; - } - await mem0.update(target.id, { text: `[PINNED] ${text}` }); - pinned(target); - return; - } - - const labels = memories.map((m) => formatMemoryCompact(m)); - const selected = await ctx.ui.select( - `Found ${pluralize(memories.length, "match", "matches")} for "${query}" — which should I pin?`, - labels, - ); - if (!selected) { - sendFeedback("mem0-pin", "**Cancelled** — nothing was pinned."); - return; - } - const idx = labels.indexOf(selected); - if (idx < 0) return; - const target = memories[idx]; - const selectedText = target.memory ?? ""; - if (selectedText.startsWith("[PINNED]")) { - alreadyPinned(target); - return; - } - await mem0.update(target.id, { text: `[PINNED] ${selectedText}` }); - pinned(target); - }, - }); - pi.registerCommand("mem0-scope", { description: "Change default memory scope for this session (project, session, global)", handler: async (args, ctx) => { @@ -329,7 +242,6 @@ export function registerCommands( `- Search relevance threshold: ${config.searchThreshold}`, `- Project memories: ${count}`, `- Auto-capture: ${config.autoCapture ? "on" : "off"}`, - `- Dream: ${config.dream.enabled ? "enabled" : "disabled"}`, ]; captureCommandEvent("mem0-status", { connected, memory_count: count }, telemetryCtx); diff --git a/integrations/pi-agent-plugin/src/config/index.ts b/integrations/pi-agent-plugin/src/config/index.ts index 8788896cf..36746b50e 100644 --- a/integrations/pi-agent-plugin/src/config/index.ts +++ b/integrations/pi-agent-plugin/src/config/index.ts @@ -1,20 +1,12 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import type { Mem0Config, DreamConfig } from "../types.ts"; +import type { Mem0Config } from "../types.ts"; const AGENT_ROOT = path.join(os.homedir(), ".pi", "agent"); export const CONFIG_DIR = AGENT_ROOT; const CONFIG_PATH = path.join(AGENT_ROOT, "mem0-config.json"); -const DEFAULT_DREAM: DreamConfig = { - enabled: true, - auto: true, - minHours: 24, - minSessions: 5, - minMemories: 20, -}; - const DEFAULT_CONFIG: Mem0Config = { apiKey: "", userId: "", @@ -22,7 +14,6 @@ const DEFAULT_CONFIG: Mem0Config = { defaultScope: "project", contextInjection: true, searchThreshold: 0.3, - dream: DEFAULT_DREAM, }; export function loadConfig(): Mem0Config { @@ -37,15 +28,9 @@ export function loadConfig(): Mem0Config { } } - const dream: DreamConfig = { - ...DEFAULT_DREAM, - ...(fileConfig.dream ?? {}), - }; - const config: Mem0Config = { ...DEFAULT_CONFIG, ...fileConfig, - dream, }; if (process.env.MEM0_API_KEY) { diff --git a/integrations/pi-agent-plugin/src/dream/index.ts b/integrations/pi-agent-plugin/src/dream/index.ts deleted file mode 100644 index 2129d947b..000000000 --- a/integrations/pi-agent-plugin/src/dream/index.ts +++ /dev/null @@ -1,115 +0,0 @@ -import * as fs from "node:fs"; -import * as path from "node:path"; -import type { DreamState, DreamLock, DreamConfig } from "../types.ts"; - -const LOCK_STALE_MS = 60 * 60 * 1000; - -const DEFAULTS: DreamConfig = { - enabled: true, - auto: true, - minHours: 24, - minSessions: 5, - minMemories: 20, -}; - -function statePath(stateDir: string): string { - return path.join(stateDir, "mem0-dream-state.json"); -} - -function lockPath(stateDir: string): string { - return path.join(stateDir, "mem0-dream.lock"); -} - -function ensureDir(dir: string): void { - try { - fs.mkdirSync(dir, { recursive: true }); - } catch { /* exists */ } -} - -function readState(stateDir: string): DreamState { - try { - const raw = fs.readFileSync(statePath(stateDir), "utf-8"); - return JSON.parse(raw) as DreamState; - } catch { - return { lastConsolidatedAt: 0, sessionsSince: 0, lastSessionId: null }; - } -} - -function writeState(stateDir: string, state: DreamState): void { - ensureDir(stateDir); - fs.writeFileSync(statePath(stateDir), JSON.stringify(state, null, 2)); -} - -export function incrementSessionCount(stateDir: string, sessionId: string): void { - const state = readState(stateDir); - if (state.lastSessionId !== sessionId) { - state.sessionsSince++; - state.lastSessionId = sessionId; - writeState(stateDir, state); - } -} - -export function checkCheapGates( - stateDir: string, - config: Partial, -): { proceed: boolean; reason?: string } { - const minHours = config.minHours ?? DEFAULTS.minHours; - const minSessions = config.minSessions ?? DEFAULTS.minSessions; - const state = readState(stateDir); - - const hoursSince = (Date.now() - state.lastConsolidatedAt) / 3_600_000; - if (hoursSince < minHours) { - return { proceed: false, reason: `time: ${hoursSince.toFixed(1)}h < ${minHours}h` }; - } - - if (state.sessionsSince < minSessions) { - return { proceed: false, reason: `sessions: ${state.sessionsSince} < ${minSessions}` }; - } - - return { proceed: true }; -} - -export function checkMemoryGate( - memoryCount: number, - config: Partial, -): { pass: boolean; reason?: string } { - const minMemories = config.minMemories ?? DEFAULTS.minMemories; - if (memoryCount < minMemories) { - return { pass: false, reason: `memories: ${memoryCount} < ${minMemories}` }; - } - return { pass: true }; -} - -export function acquireDreamLock(stateDir: string): boolean { - ensureDir(stateDir); - const lp = lockPath(stateDir); - - try { - const raw = fs.readFileSync(lp, "utf-8"); - const lock = JSON.parse(raw) as DreamLock; - if (Date.now() - lock.startedAt < LOCK_STALE_MS) { - return false; - } - try { fs.unlinkSync(lp); } catch { /* race ok */ } - } catch { /* no lock file */ } - - const lock: DreamLock = { pid: process.pid, startedAt: Date.now() }; - try { - fs.writeFileSync(lp, JSON.stringify(lock), { flag: "wx" }); - return true; - } catch { - return false; - } -} - -export function releaseDreamLock(stateDir: string): void { - try { fs.unlinkSync(lockPath(stateDir)); } catch { /* already gone */ } -} - -export function recordDreamCompletion(stateDir: string): void { - const state = readState(stateDir); - state.lastConsolidatedAt = Date.now(); - state.sessionsSince = 0; - state.lastSessionId = null; - writeState(stateDir, state); -} diff --git a/integrations/pi-agent-plugin/src/dream/prompt.ts b/integrations/pi-agent-plugin/src/dream/prompt.ts deleted file mode 100644 index 4da1e7171..000000000 --- a/integrations/pi-agent-plugin/src/dream/prompt.ts +++ /dev/null @@ -1,22 +0,0 @@ -export const DREAM_PROTOCOL = ` -You are running memory consolidation. Complete these steps using the mem0_memory tool: - -1. ORIENT — Call mem0_memory with action "get_all" to list all memories. Count by category. Note oldest/newest. - -2. GATHER TARGETS — Review each memory. Classify as: - - DELETE: sensitive information (API keys, passwords, tokens), expired/stale entries, noise, redundant operational details - - MERGE: near-duplicates (same fact stated differently). Keep the better-worded one, delete the other. - - REWRITE: vague, first-person, or poorly-categorized entries. Use mem0_memory "add" with improved text, then "delete" the old one. - - KEEP: everything else. - Skip any memory starting with "[PINNED]". - -3. CONSOLIDATE — Execute the changes: - - Delete stale/duplicate entries - - For merges: add the merged text, delete both originals - - For rewrites: add improved version, delete original - -4. REPORT — Summarize: how many reviewed, deleted, merged, rewritten, final count. - -Quality targets: zero sensitive data stored, zero duplicates, all entries are atomic (one fact each), 15-50 words each. -After consolidation, respond to the user's message normally. -`; diff --git a/integrations/pi-agent-plugin/src/entry.ts b/integrations/pi-agent-plugin/src/entry.ts index 51ad5bcd0..d49d78ad0 100644 --- a/integrations/pi-agent-plugin/src/entry.ts +++ b/integrations/pi-agent-plugin/src/entry.ts @@ -1,24 +1,17 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; import MemoryClient from "mem0ai"; -import { loadConfig, CONFIG_DIR } from "./config/index.ts"; +import { loadConfig } from "./config/index.ts"; import { detectAppId, detectRunId, resolveSearchFilters } from "./memory/scoping.ts"; -import { formatMemoryList } from "./memory/formatting.ts"; import { registerMemoryTool } from "./memory/tools.ts"; import { registerCommands } from "./commands.ts"; import { setupAutoCapture } from "./capture/index.ts"; import { MEMORY_POLICY } from "./prompt.ts"; -import { DREAM_PROTOCOL } from "./dream/prompt.ts"; -import { - incrementSessionCount, - checkCheapGates, - checkMemoryGate, - acquireDreamLock, - releaseDreamLock, - recordDreamCompletion, -} from "./dream/index.ts"; import { captureEvent } from "./telemetry.ts"; import * as os from "node:os"; import type { ScopeContext } from "./types.ts"; +import { createMemoryLifecycle } from "../../agent-plugin-core/typescript/src/lifecycle.ts"; + +export { buildRecallContext } from "../../agent-plugin-core/typescript/src/lifecycle.ts"; export function resolveUserId(configUserId: string): string { if (configUserId) return configUserId; @@ -27,31 +20,6 @@ export function resolveUserId(configUserId: string): string { try { return os.userInfo().username; } catch { return "default"; } } -/** - * Build the auto-recall context block for a turn: search memory with the user's - * prompt and format the top matches so they are guaranteed in context instead of - * relying on the agent to call the tool. Best-effort — returns "" when disabled, - * the prompt is blank, nothing matches, or the search fails; it must never block - * the turn. - */ -export async function buildRecallContext( - prompt: string, - enabled: boolean, - search: (query: string) => Promise<{ results?: unknown[] }>, -): Promise { - if (!enabled) return ""; - const q = prompt.trim(); - if (!q) return ""; - try { - const res = await search(q); - const memories = (res.results ?? []) as Parameters[0]; - if (memories.length === 0) return ""; - return `\nRetrieved automatically for the current request. This is a shallow first pass — search mem0_memory for more if you need it.\n${formatMemoryList(memories)}\n`; - } catch { - return ""; - } -} - export default function mem0Extension(pi: ExtensionAPI): void { const config = loadConfig(); @@ -73,20 +41,21 @@ export default function mem0Extension(pi: ExtensionAPI): void { } const telemetryCtx = { apiKey: config.apiKey }; + const lifecycle = createMemoryLifecycle(); // ── Register tool + commands + auto-capture ───────────────────────── registerMemoryTool(pi, mem0, config, getScopeCtx, telemetryCtx); registerCommands(pi, mem0, config, getScopeCtx, telemetryCtx); - setupAutoCapture(pi, mem0, config, getScopeCtx, telemetryCtx); + setupAutoCapture(pi, mem0, config, getScopeCtx, telemetryCtx, lifecycle); captureEvent("pi.plugin.registered", { auto_capture: config.autoCapture, - dream_enabled: config.dream.enabled, default_scope: config.defaultScope, }, telemetryCtx); // ── session_start: detect project + session, reconstruct scope ────── pi.on("session_start", async (_event, ctx) => { + lifecycle.beginSession(); scopeCtx.appId = detectAppId(ctx.cwd); const sessionFile = ctx.sessionManager?.getSessionFile?.(); @@ -96,86 +65,29 @@ export default function mem0Extension(pi: ExtensionAPI): void { scopeCtx.userId = config.userId; } - if (config.dream.enabled) { - incrementSessionCount(CONFIG_DIR, scopeCtx.runId); - } - captureEvent("pi.session.start", {}, telemetryCtx); }); - // ── before_agent_start: append memory policy + auto-dream trigger ─── - let dreamTriggered = false; - let dreamChecked = false; - + // ── before_agent_start: append memory policy and recall ───────────── pi.on("before_agent_start", async (event, _ctx) => { let extra = MEMORY_POLICY; // Guaranteed retrieval: prefetch memories relevant to this prompt so the // agent always has them, rather than depending on it to call the tool. - const recall = await buildRecallContext( + const recall = await lifecycle.recall( event.prompt ?? "", config.contextInjection, (q) => mem0.search(q, { filters: resolveSearchFilters("project", scopeCtx) }), ); if (recall) extra += "\n\n" + recall; - if (config.dream.enabled && config.dream.auto && !dreamTriggered && !dreamChecked) { - const gates = checkCheapGates(CONFIG_DIR, config.dream); - if (gates.proceed) { - try { - const filters = resolveSearchFilters("project", scopeCtx); - const result = await mem0.getAll({ filters }); - const count = result.count ?? (result.results ?? []).length; - dreamChecked = true; - const memGate = checkMemoryGate(count, config.dream); - - if (memGate.pass && acquireDreamLock(CONFIG_DIR)) { - dreamTriggered = true; - extra += "\n\n" + DREAM_PROTOCOL; - captureEvent("pi.dream.triggered", { memory_count: count }, telemetryCtx); - } - } catch { - // Transient error — retry next turn - } - } - } - return { systemPrompt: (event.systemPrompt ?? "") + "\n\n" + extra, }; }); - // ── agent_end: dream completion check ─────────────────────────────── - pi.on("agent_end", async (event) => { - if (!dreamTriggered) return; - - const messages = event.messages ?? []; - const hadWriteAction = messages.some((m) => { - if (m.role !== "assistant") return false; - const content = Array.isArray(m.content) ? m.content : []; - return content.some( - (block: any) => - block.type === "tool_use" && - block.name === "mem0_memory" && - ["add", "delete", "delete_all"].includes(block.input?.action), - ); - }); - - if (hadWriteAction) { - recordDreamCompletion(CONFIG_DIR); - captureEvent("pi.dream.completed", {}, telemetryCtx); - } - - releaseDreamLock(CONFIG_DIR); - dreamTriggered = false; - }); - - // ── session_shutdown: release dream lock if still held ────────────── + // ── session_shutdown ──────────────────────────────────────────────── pi.on("session_shutdown", async () => { captureEvent("pi.session.stop", {}, telemetryCtx); - if (dreamTriggered) { - releaseDreamLock(CONFIG_DIR); - dreamTriggered = false; - } }); } diff --git a/integrations/pi-agent-plugin/src/index.ts b/integrations/pi-agent-plugin/src/index.ts index 0da362d47..228ef3833 100644 --- a/integrations/pi-agent-plugin/src/index.ts +++ b/integrations/pi-agent-plugin/src/index.ts @@ -1,7 +1,6 @@ export type { Scope, Mem0Config, - DreamConfig, ScopeContext, CustomCategory, } from "./types.ts"; @@ -15,16 +14,6 @@ export { formatAge, formatMemoryCompact, formatMemoryList, groupByCategory } fro export { setupAutoCapture, extractConversation } from "./capture/index.ts"; -export { - incrementSessionCount, - checkCheapGates, - checkMemoryGate, - acquireDreamLock, - releaseDreamLock, - recordDreamCompletion, -} from "./dream/index.ts"; -export { DREAM_PROTOCOL } from "./dream/prompt.ts"; - export { MEMORY_POLICY } from "./prompt.ts"; export { registerCommands } from "./commands.ts"; diff --git a/integrations/pi-agent-plugin/src/memory/formatting.ts b/integrations/pi-agent-plugin/src/memory/formatting.ts index 5d6bd3366..e8f4c6461 100644 --- a/integrations/pi-agent-plugin/src/memory/formatting.ts +++ b/integrations/pi-agent-plugin/src/memory/formatting.ts @@ -1,43 +1,7 @@ -interface MemoryLike { - id: string; - memory?: string; - categories?: string[]; - createdAt?: Date | string; -} - -export function formatAge(date: Date | string): string { - const d = typeof date === "string" ? new Date(date) : date; - const ms = Date.now() - d.getTime(); - const minutes = Math.floor(ms / 60_000); - if (minutes < 60) return `${minutes}m ago`; - const hours = Math.floor(minutes / 60); - if (hours < 24) return `${hours}h ago`; - const days = Math.floor(hours / 24); - return `${days}d ago`; -} - -export function formatMemoryCompact(mem: MemoryLike): string { - const cat = mem.categories?.[0] ?? "uncategorized"; - const age = mem.createdAt ? ` (${formatAge(mem.createdAt)})` : ""; - return `[${cat}] ${mem.memory ?? "(empty)"}${age} [mem0:${mem.id}]`; -} - -export function formatMemoryList(memories: MemoryLike[]): string { - if (memories.length === 0) return "No memories found."; - return memories - .map((m, i) => `${i + 1}. ${formatMemoryCompact(m)}`) - .join("\n"); -} - -export function groupByCategory( - memories: MemoryLike[], -): Map { - const groups = new Map(); - for (const m of memories) { - const cat = m.categories?.[0] ?? "uncategorized"; - const list = groups.get(cat) ?? []; - list.push(m); - groups.set(cat, list); - } - return groups; -} +export { + formatAge, + formatMemoryCompact, + formatMemoryList, + groupByCategory, +} from "../../../agent-plugin-core/typescript/src/formatting.ts"; +export type { MemoryLike } from "../../../agent-plugin-core/typescript/src/formatting.ts"; diff --git a/integrations/pi-agent-plugin/src/memory/scoping.test.ts b/integrations/pi-agent-plugin/src/memory/scoping.test.ts index ee843a176..35229ee3c 100644 --- a/integrations/pi-agent-plugin/src/memory/scoping.test.ts +++ b/integrations/pi-agent-plugin/src/memory/scoping.test.ts @@ -63,8 +63,8 @@ describe("resolveSearchFilters", () => { expect(resolveSearchFilters("session", ctx)).toEqual({ user_id: "u1", app_id: "a1", run_id: "r1" }); }); - it("uses wildcard app_id for global scope", () => { - expect(resolveSearchFilters("global", ctx)).toEqual({ user_id: "u1", app_id: "*" }); + it("limits global scope to the configured user", () => { + expect(resolveSearchFilters("global", ctx)).toEqual({ user_id: "u1" }); }); }); diff --git a/integrations/pi-agent-plugin/src/memory/scoping.ts b/integrations/pi-agent-plugin/src/memory/scoping.ts index 8f5d2af44..3131ab57c 100644 --- a/integrations/pi-agent-plugin/src/memory/scoping.ts +++ b/integrations/pi-agent-plugin/src/memory/scoping.ts @@ -2,6 +2,10 @@ import * as path from "node:path"; import * as crypto from "node:crypto"; import { execFileSync } from "node:child_process"; import type { Scope, ScopeContext } from "../types.ts"; +import { + scopeAddParams, + scopeSearchFilters, +} from "../../../agent-plugin-core/typescript/src/scoping.ts"; export function detectAppId(cwd: string): string { try { @@ -26,26 +30,12 @@ export function resolveSearchFilters( scope: Scope, ctx: ScopeContext, ): Record { - switch (scope) { - case "project": - return { user_id: ctx.userId, app_id: ctx.appId }; - case "session": - return { user_id: ctx.userId, app_id: ctx.appId, run_id: ctx.runId }; - case "global": - return { user_id: ctx.userId, app_id: "*" }; - } + return scopeSearchFilters(scope, ctx); } export function resolveAddParams( scope: Scope, ctx: ScopeContext, ): Record { - switch (scope) { - case "project": - return { userId: ctx.userId, appId: ctx.appId }; - case "session": - return { userId: ctx.userId, appId: ctx.appId, runId: ctx.runId }; - case "global": - return { userId: ctx.userId }; - } + return scopeAddParams(scope, ctx); } diff --git a/integrations/pi-agent-plugin/src/memory/tools.ts b/integrations/pi-agent-plugin/src/memory/tools.ts index 0ab06c142..96d8e9b3e 100644 --- a/integrations/pi-agent-plugin/src/memory/tools.ts +++ b/integrations/pi-agent-plugin/src/memory/tools.ts @@ -1,3 +1,4 @@ +import { resolveToolScope } from "../../../agent-plugin-core/typescript/src/scoping.ts"; import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; import { Type } from "typebox"; import { StringEnum } from "@earendil-works/pi-ai"; @@ -17,6 +18,10 @@ interface MemoryResult { const MAX_OUTPUT_LINES = 200; const MAX_OUTPUT_BYTES = 50_000; +function normalizeMemoryId(id: string): string { + return id.replace(/^\[?mem0:([0-9a-f-]{36})\]?$/i, "$1"); +} + function truncateOutput(text: string): string { const lines = text.split("\n"); if (lines.length <= MAX_OUTPUT_LINES && text.length <= MAX_OUTPUT_BYTES) { @@ -50,7 +55,7 @@ export function buildToolExecute( defaultScope: Scope, ) { return async (params: ToolParams, signal?: AbortSignal) => { - const scope = params.scope ?? defaultScope; + const scope = resolveToolScope(params.scope, defaultScope); switch (params.action) { case "search": { @@ -96,18 +101,19 @@ export function buildToolExecute( if (signal?.aborted) throw new Error("Cancelled"); if (!params.memory_id) throw new Error("memory_id is required for update"); if (!params.content) throw new Error("content is required for update"); - const updateResult = await mem0.update(params.memory_id, { text: params.content }); + const memoryId = normalizeMemoryId(params.memory_id); + const updateResult = await mem0.update(memoryId, { text: params.content }); const res = updateResult as MemoryResult; return { content: [{ type: "text" as const, text: res.status ?? "Memory updated." }], - details: { memoryId: params.memory_id }, + details: { memoryId }, }; } case "delete": { if (signal?.aborted) throw new Error("Cancelled"); if (!params.memory_id) throw new Error("memory_id is required for delete"); - const result = await mem0.delete(params.memory_id); + const result = await mem0.delete(normalizeMemoryId(params.memory_id)); return { content: [{ type: "text" as const, text: result.message ?? "Memory deleted." }], details: {}, @@ -145,7 +151,7 @@ export function registerMemoryTool( 'For multi-part or comparative questions, run several searches with different phrasings and combine the results before answering -- one search is rarely enough', 'Use mem0_memory with action "add" to save important facts, preferences, goals, decisions, or lessons the user shares', 'Use mem0_memory with action "update" to modify an existing memory — requires memory_id and content. Preserves the memory ID', - "Always use the default project scope unless the user EXPLICITLY asks to search across all projects — only then use scope \"global\"", + "Always use the default project scope unless the user EXPLICITLY asks to search across all projects — only after the user selects /mem0-scope global use scope \"global\"", "Do NOT pass scope at all for normal queries — omitting it uses the project default automatically", ], parameters: Type.Object({ diff --git a/integrations/pi-agent-plugin/src/telemetry.ts b/integrations/pi-agent-plugin/src/telemetry.ts index f9c5eb568..fa98ddf18 100644 --- a/integrations/pi-agent-plugin/src/telemetry.ts +++ b/integrations/pi-agent-plugin/src/telemetry.ts @@ -1,240 +1,99 @@ -/** - * Plugin telemetry — anonymous usage tracking via PostHog. - * - * Sends fire-and-forget events to PostHog using native fetch(). - * Events are batched and flushed every 5 seconds or when the queue - * reaches 10 events, whichever comes first. - * - * Disable with: MEM0_TELEMETRY=false - */ - import { createHash, randomUUID } from "node:crypto"; import * as fs from "node:fs"; import * as path from "node:path"; + +import { createTelemetry } from "../../agent-plugin-core/typescript/src/telemetry.ts"; import { CONFIG_DIR } from "./config/index.ts"; -const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"; -const POSTHOG_HOST = "https://us.i.posthog.com/i/v0/e/"; - -const FLUSH_INTERVAL_MS = 5_000; -const FLUSH_THRESHOLD = 10; - -let eventQueue: Record[] = []; -let flushTimer: ReturnType | undefined; - -function _loadPluginVersion(): string { +const PLUGIN_VERSION = (() => { try { - const pkgUrl = new URL("../package.json", import.meta.url); - const pkg = JSON.parse(fs.readFileSync(pkgUrl, "utf-8")); - return pkg.version ?? "unknown"; + return JSON.parse(fs.readFileSync(new URL("../package.json", import.meta.url), "utf-8")).version; } catch { return "unknown"; } -} - -const PLUGIN_VERSION = _loadPluginVersion(); - -// ── Opt-out ────────────────────────────────────────────────────────────── - -function isTelemetryEnabled(): boolean { - try { - const val = process.env.MEM0_TELEMETRY; - if (val !== undefined) { - const s = val.toLowerCase(); - return s !== "false" && s !== "0" && s !== "no" && s !== "off"; - } - return true; - } catch { - return true; - } -} - -// ── Identity ───────────────────────────────────────────────────────────── - +})(); const TELEMETRY_ID_PATH = path.join(CONFIG_DIR, "mem0-telemetry-id.json"); -let _cachedAnonymousId: string | undefined; +let cachedAnonymousId: string | undefined; +let currentDistinctId = ""; +let identified = false; -function getOrCreateAnonymousId(): string { - if (_cachedAnonymousId) return _cachedAnonymousId; +function anonymousId(): string { + if (cachedAnonymousId) return cachedAnonymousId; try { - if (fs.existsSync(TELEMETRY_ID_PATH)) { - const data = JSON.parse(fs.readFileSync(TELEMETRY_ID_PATH, "utf-8")); - if (data.anonymousId) { - _cachedAnonymousId = data.anonymousId; - return _cachedAnonymousId!; - } - } - } catch { /* ignore */ } - - const newId = `pi-mem0-anon-${randomUUID().replace(/-/g, "")}`; + const stored = JSON.parse(fs.readFileSync(TELEMETRY_ID_PATH, "utf-8")).anonymousId; + if (typeof stored === "string" && stored) return (cachedAnonymousId = stored); + } catch { + // First run or unreadable identity file. + } + const created = `pi-mem0-anon-${randomUUID().replace(/-/g, "")}`; try { fs.mkdirSync(CONFIG_DIR, { recursive: true }); - fs.writeFileSync(TELEMETRY_ID_PATH, JSON.stringify({ anonymousId: newId }), "utf-8"); - } catch { /* ignore */ } - _cachedAnonymousId = newId; - return newId; -} - -function getDistinctId(apiKey?: string): string { - if (apiKey) { - return createHash("sha256").update(apiKey).digest("hex"); - } - return getOrCreateAnonymousId(); -} - -let _identifyDone = false; - -function maybeBuildIdentifyEvent(distinctId: string): Record | null { - if (_identifyDone) return null; - if (!distinctId || distinctId.startsWith("pi-mem0-anon-")) return null; - try { - if (!fs.existsSync(TELEMETRY_ID_PATH)) { - _identifyDone = true; - return null; - } - const data = JSON.parse(fs.readFileSync(TELEMETRY_ID_PATH, "utf-8")); - const storedAnon = data.anonymousId; - if (!storedAnon) { - _identifyDone = true; - return null; - } - const identifyEvent = { - event: "$identify", - distinct_id: distinctId, - properties: { $anon_distinct_id: storedAnon, $lib: "posthog-node" }, - }; - try { - fs.unlinkSync(TELEMETRY_ID_PATH); - } catch { /* ignore */ } - _identifyDone = true; - _cachedAnonymousId = undefined; - return identifyEvent; + fs.writeFileSync(TELEMETRY_ID_PATH, JSON.stringify({ anonymousId: created }), "utf-8"); } catch { - return null; + // An unwritable config directory must not break the plugin. + } + return (cachedAnonymousId = created); +} + +function distinctId(apiKey?: string): string { + return apiKey ? createHash("sha256").update(apiKey).digest("hex") : anonymousId(); +} + +function previousAnonymousId(id: string): string | undefined { + if (identified || id.startsWith("pi-mem0-anon-")) return undefined; + identified = true; + try { + const stored = JSON.parse(fs.readFileSync(TELEMETRY_ID_PATH, "utf-8")).anonymousId; + fs.unlinkSync(TELEMETRY_ID_PATH); + cachedAnonymousId = undefined; + return typeof stored === "string" && stored ? stored : undefined; + } catch { + return undefined; } } -// ── Flush machinery ────────────────────────────────────────────────────── - -function ensureFlushTimer(): void { - if (flushTimer) return; - flushTimer = setInterval(flushEvents, FLUSH_INTERVAL_MS); - if (typeof flushTimer === "object" && "unref" in flushTimer) { - flushTimer.unref(); - } -} - -let _exitHandlerInstalled = false; - -function ensureExitHandler(): void { - if (_exitHandlerInstalled) return; - _exitHandlerInstalled = true; - process.on("beforeExit", async () => { - if (eventQueue.length === 0) return; - const batch = eventQueue; - eventQueue = []; - const body = JSON.stringify({ api_key: POSTHOG_API_KEY, batch }); - try { - await fetch(POSTHOG_HOST, { - method: "POST", - headers: { - "Content-Type": "application/json", - "Content-Length": String(Buffer.byteLength(body)), - }, - body, - signal: AbortSignal.timeout(3_000), - }); - } catch { /* silently swallow */ } - }); -} - -function flushEvents(): void { - if (eventQueue.length === 0) return; - const batch = eventQueue; - eventQueue = []; - - const body = JSON.stringify({ api_key: POSTHOG_API_KEY, batch }); - fetch(POSTHOG_HOST, { - method: "POST", - headers: { - "Content-Type": "application/json", - "Content-Length": String(Buffer.byteLength(body)), - }, - body, - signal: AbortSignal.timeout(3_000), - }).catch(() => { /* silently swallow */ }); -} - -// ── Public API ─────────────────────────────────────────────────────────── +const telemetry = createTelemetry({ + host: "pi", + source: "PI_AGENT_PLUGIN", + version: PLUGIN_VERSION, + distinctId: () => currentDistinctId, +}); export function captureEvent( eventName: string, properties: Record = {}, - ctx?: { apiKey?: string }, + context?: { apiKey?: string }, ): void { - if (!isTelemetryEnabled()) return; - - try { - const distinctId = getDistinctId(ctx?.apiKey); - - const identifyEvent = maybeBuildIdentifyEvent(distinctId); - if (identifyEvent) { - eventQueue.push(identifyEvent); - } - - eventQueue.push({ - event: eventName, - distinct_id: distinctId, - properties: { - source: "PI_AGENT_PLUGIN", - language: "node", - plugin_version: PLUGIN_VERSION, - node_version: process.version, - os: process.platform, - $process_person_profile: false, - $lib: "posthog-node", - ...properties, - }, - }); - - ensureFlushTimer(); - ensureExitHandler(); - - if (eventQueue.length >= FLUSH_THRESHOLD) { - flushEvents(); - } - } catch { /* silently swallow */ } + currentDistinctId = distinctId(context?.apiKey); + const anonymous = previousAnonymousId(currentDistinctId); + if (anonymous) telemetry.capture("$identify", { $anon_distinct_id: anonymous }); + telemetry.capture(eventName, properties); } export function captureToolEvent( action: string, properties: Record = {}, - ctx?: { apiKey?: string }, + context?: { apiKey?: string }, ): void { - captureEvent("pi.tool.mem0_memory", { action, ...properties }, ctx); + captureEvent("pi.tool.mem0_memory", { action, ...properties }, context); } export function captureCommandEvent( command: string, properties: Record = {}, - ctx?: { apiKey?: string }, + context?: { apiKey?: string }, ): void { - captureEvent(`pi.command.${command}`, properties, ctx); + captureEvent(`pi.command.${command}`, properties, context); } -// ── Test helpers ───────────────────────────────────────────────────────── - export function _getEventQueue(): Record[] { - return eventQueue; + return telemetry.queueForTesting(); } export function _resetForTesting(): void { - eventQueue = []; - if (flushTimer) { - clearInterval(flushTimer); - flushTimer = undefined; - } - _cachedAnonymousId = undefined; - _identifyDone = false; + telemetry.resetForTesting(); + cachedAnonymousId = undefined; + currentDistinctId = ""; + identified = false; } diff --git a/integrations/pi-agent-plugin/src/types.ts b/integrations/pi-agent-plugin/src/types.ts index e23e80e4c..c2243cb37 100644 --- a/integrations/pi-agent-plugin/src/types.ts +++ b/integrations/pi-agent-plugin/src/types.ts @@ -1,13 +1,5 @@ export type Scope = "project" | "session" | "global"; -export interface DreamConfig { - enabled: boolean; - auto: boolean; - minHours: number; - minSessions: number; - minMemories: number; -} - export interface Mem0Config { apiKey: string; userId: string; @@ -15,18 +7,6 @@ export interface Mem0Config { defaultScope: Scope; contextInjection: boolean; searchThreshold: number; - dream: DreamConfig; -} - -export interface DreamState { - lastConsolidatedAt: number; - sessionsSince: number; - lastSessionId: string | null; -} - -export interface DreamLock { - pid: number; - startedAt: number; } export interface ScopeContext { diff --git a/integrations/pi-agent-plugin/tests/capture.test.ts b/integrations/pi-agent-plugin/tests/capture.test.ts index 6f07b3035..5651909be 100644 --- a/integrations/pi-agent-plugin/tests/capture.test.ts +++ b/integrations/pi-agent-plugin/tests/capture.test.ts @@ -59,4 +59,10 @@ describe("extractConversation", () => { it("returns empty array for empty input", () => { expect(extractConversation([])).toEqual([]); }); + + it("redacts credentials before automatic capture", () => { + expect(extractConversation([ + { role: "user", content: "api_key=do-not-store-this" }, + ])).toEqual([{ role: "user", content: "api_key=[REDACTED]" }]); + }); }); diff --git a/integrations/pi-agent-plugin/tests/config.test.ts b/integrations/pi-agent-plugin/tests/config.test.ts index 434210938..524cbd2ef 100644 --- a/integrations/pi-agent-plugin/tests/config.test.ts +++ b/integrations/pi-agent-plugin/tests/config.test.ts @@ -43,8 +43,6 @@ describe("loadConfig", () => { const config = loadConfig(); expect(config.apiKey).toBe("m0-file-key"); expect(config.userId).toBe("file-user"); - expect(config.dream.enabled).toBe(true); - expect(config.dream.minHours).toBe(24); }); it("env vars override config file", () => { diff --git a/integrations/pi-agent-plugin/tests/dream.test.ts b/integrations/pi-agent-plugin/tests/dream.test.ts deleted file mode 100644 index 42c40223c..000000000 --- a/integrations/pi-agent-plugin/tests/dream.test.ts +++ /dev/null @@ -1,59 +0,0 @@ -import { describe, it, expect, vi, beforeEach, afterEach } from "vitest"; -import * as fs from "node:fs"; -import { checkCheapGates, checkMemoryGate } from "../src/dream/index.ts"; - -vi.mock("node:fs"); - -const STATE_DIR = "/tmp/test-mem0-dream"; - -describe("checkCheapGates", () => { - beforeEach(() => { - vi.mocked(fs.mkdirSync).mockReturnValue(undefined); - }); - - afterEach(() => { - vi.restoreAllMocks(); - }); - - it("blocks when no state file exists (zero sessions)", () => { - vi.mocked(fs.readFileSync).mockImplementation(() => { throw new Error("ENOENT"); }); - const result = checkCheapGates(STATE_DIR, {}); - expect(result.proceed).toBe(false); - expect(result.reason).toContain("sessions"); - }); - - it("blocks when consolidated recently", () => { - vi.mocked(fs.readFileSync).mockReturnValue( - JSON.stringify({ lastConsolidatedAt: Date.now() - 3_600_000, sessionsSince: 10, lastSessionId: null }) - ); - const result = checkCheapGates(STATE_DIR, {}); - expect(result.proceed).toBe(false); - expect(result.reason).toContain("time"); - }); - - it("blocks when not enough sessions", () => { - vi.mocked(fs.readFileSync).mockReturnValue( - JSON.stringify({ lastConsolidatedAt: Date.now() - 48 * 3_600_000, sessionsSince: 2, lastSessionId: null }) - ); - const result = checkCheapGates(STATE_DIR, {}); - expect(result.proceed).toBe(false); - expect(result.reason).toContain("sessions"); - }); - - it("proceeds when both gates pass", () => { - vi.mocked(fs.readFileSync).mockReturnValue( - JSON.stringify({ lastConsolidatedAt: Date.now() - 48 * 3_600_000, sessionsSince: 10, lastSessionId: null }) - ); - expect(checkCheapGates(STATE_DIR, {}).proceed).toBe(true); - }); -}); - -describe("checkMemoryGate", () => { - it("blocks when too few memories", () => { - expect(checkMemoryGate(5, {}).pass).toBe(false); - }); - - it("passes when enough memories", () => { - expect(checkMemoryGate(25, {}).pass).toBe(true); - }); -}); diff --git a/integrations/pi-agent-plugin/tests/scoping.test.ts b/integrations/pi-agent-plugin/tests/scoping.test.ts index 9cf468216..d5b34a531 100644 --- a/integrations/pi-agent-plugin/tests/scoping.test.ts +++ b/integrations/pi-agent-plugin/tests/scoping.test.ts @@ -21,9 +21,9 @@ describe("resolveSearchFilters", () => { }); }); - it("global scope returns user_id with app_id wildcard", () => { + it("global scope returns user_id only", () => { expect(resolveSearchFilters("global", ctx)).toEqual({ - user_id: "kartik", app_id: "*", + user_id: "kartik", }); }); }); diff --git a/integrations/pi-agent-plugin/tests/tools.test.ts b/integrations/pi-agent-plugin/tests/tools.test.ts index f9b753ef5..5ea892d81 100644 --- a/integrations/pi-agent-plugin/tests/tools.test.ts +++ b/integrations/pi-agent-plugin/tests/tools.test.ts @@ -6,6 +6,7 @@ const mockMem0 = { search: vi.fn(), add: vi.fn(), getAll: vi.fn(), + update: vi.fn(), delete: vi.fn(), deleteAll: vi.fn(), }; @@ -38,11 +39,13 @@ describe("buildToolExecute", () => { expect(call[1].customCategories.length).toBe(10); }); - it("search with scope=global filters by user_id with app_id wildcard", async () => { + it("search uses global scope only after the user selects it", async () => { mockMem0.search.mockResolvedValue({ results: [] }); - await execute({ action: "search", query: "preferences", scope: "global" }); + await expect(execute({ action: "search", query: "preferences", scope: "global" })).rejects.toThrow(/Select global/); + const globalExecute = buildToolExecute(mockMem0 as any, scopeCtx, "global"); + await globalExecute({ action: "search", query: "preferences", scope: "global" }); expect(mockMem0.search).toHaveBeenCalledWith("preferences", { - filters: { user_id: "testuser", app_id: "*" }, + filters: { user_id: "testuser" }, }); }); @@ -58,4 +61,19 @@ describe("buildToolExecute", () => { await execute({ action: "delete", memory_id: fullId }); expect(mockMem0.delete).toHaveBeenCalledWith(fullId); }); + + it("write actions accept the citation ID returned by search", async () => { + mockMem0.update.mockResolvedValue({ status: "updated" }); + mockMem0.delete.mockResolvedValue({ message: "deleted" }); + await execute({ + action: "update", + memory_id: "[mem0:956e3d68-b420-4e07-a4e3-3019e7cebe6f]", + content: "updated memory", + }); + await execute({ action: "delete", memory_id: "mem0:956e3d68-b420-4e07-a4e3-3019e7cebe6f" }); + expect(mockMem0.update).toHaveBeenLastCalledWith("956e3d68-b420-4e07-a4e3-3019e7cebe6f", { + text: "updated memory", + }); + expect(mockMem0.delete).toHaveBeenLastCalledWith("956e3d68-b420-4e07-a4e3-3019e7cebe6f"); + }); }); diff --git a/integrations/pi-agent-plugin/tsconfig.json b/integrations/pi-agent-plugin/tsconfig.json index 8808fe6ed..8aa40a76d 100644 --- a/integrations/pi-agent-plugin/tsconfig.json +++ b/integrations/pi-agent-plugin/tsconfig.json @@ -8,7 +8,7 @@ "declarationMap": true, "sourceMap": true, "outDir": "dist", - "rootDir": "src", + "rootDir": "..", "strict": true, "types": ["node"], "esModuleInterop": true, diff --git a/marketplace.json b/marketplace.json index b1f765015..eea48e2af 100644 --- a/marketplace.json +++ b/marketplace.json @@ -13,7 +13,7 @@ }, "category": "Productivity", "description": "Cross-session memory and token savings for coding agents.", - "version": "0.3.0" + "version": "0.3.1" } ] } diff --git a/skills/mem0-integrate/SKILL.md b/skills/mem0-integrate/SKILL.md index 4b651d69a..34dfd284f 100644 --- a/skills/mem0-integrate/SKILL.md +++ b/skills/mem0-integrate/SKILL.md @@ -46,7 +46,7 @@ standalone `SKILL.md` with triggers, examples, and version-pinned code. - SDK (Python + TS, Platform + OSS): https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0/SKILL.md - CLI: https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0-cli/SKILL.md - Vercel AI SDK: https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0-vercel-ai-sdk/SKILL.md -- Editor/MCP plugin glue (9 MCP tools): https://github.com/mem0ai/mem0/tree/main/integrations/mem0-plugin +- Portable editor/MCP plugin: https://github.com/mem0ai/mem0/tree/main/integrations/mem0-agent-plugin ### SDK source (read when docs are ambiguous) Public repo. Cross-check against the `mem0_tested_versions` range in this @@ -109,7 +109,7 @@ the target stack. If yes, delegate — copy its call-site pattern into |---|---|---| | `@ai-sdk/*` + `ai` in `package.json` | `skills/mem0-vercel-ai-sdk` | Integration is via `createMem0` provider wrapper, not raw `MemoryClient`. | | CLI-only repo (Typer, Commander, Click, Cobra) with no LLM call sites | `skills/mem0-cli` | Call sites are command handlers, not model wrappers. Consider whether mem0 actually fits first. | -| Target is an MCP client / editor config (Claude Code, Cursor, Codex settings) | `integrations/mem0-plugin` | Wire via MCP server URL + hooks; no SDK code usually needed. | +| Target is an MCP client / editor config (Claude Code, Cursor, Codex settings) | `integrations/mem0-agent-plugin` | Wire via MCP server URL + hooks; no SDK code usually needed. | | Any other Python or TS repo with an LLM call site | `skills/mem0` | Default SDK integration path. | Record the delegated skill's raw URL in `plan.md` under a