Merge remote-tracking branch 'origin/main' into fix/telemetry-stitching
# Conflicts: # mem0-ts/src/oss/src/memory/index.ts # mem0-ts/src/oss/src/utils/telemetry.ts # mem0/memory/main.py
This commit is contained in:
@@ -8,7 +8,7 @@
|
||||
"name": "mem0",
|
||||
"source": {
|
||||
"source": "local",
|
||||
"path": "./mem0-plugin"
|
||||
"path": "./integrations/mem0-plugin"
|
||||
},
|
||||
"policy": {
|
||||
"installation": "AVAILABLE",
|
||||
|
||||
@@ -10,9 +10,9 @@
|
||||
"plugins": [
|
||||
{
|
||||
"name": "mem0",
|
||||
"source": "./mem0-plugin",
|
||||
"source": "./integrations/mem0-plugin",
|
||||
"description": "Mem0 memory layer for AI applications. Add persistent memory, personalization, and semantic search to Claude workflows.",
|
||||
"version": "0.2.9"
|
||||
"version": "0.2.10"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
"name": "mem0",
|
||||
"source": {
|
||||
"source": "local",
|
||||
"path": "./mem0-plugin"
|
||||
"path": "./integrations/mem0-plugin"
|
||||
},
|
||||
"policy": {
|
||||
"installation": "AVAILABLE",
|
||||
|
||||
@@ -10,9 +10,9 @@
|
||||
"plugins": [
|
||||
{
|
||||
"name": "mem0",
|
||||
"source": "./mem0-plugin",
|
||||
"source": "./integrations/mem0-plugin",
|
||||
"description": "Mem0 memory layer for AI applications. Add persistent memory, personalization, and semantic search.",
|
||||
"version": "0.2.9"
|
||||
"version": "0.2.10"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -68,15 +68,15 @@ jobs:
|
||||
- '.github/workflows/cli-node-ci.yml'
|
||||
- '.github/workflows/ci-gate.yml'
|
||||
openclaw:
|
||||
- 'openclaw/**'
|
||||
- 'integrations/openclaw/**'
|
||||
- '.github/workflows/openclaw-checks.yml'
|
||||
- '.github/workflows/ci-gate.yml'
|
||||
opencode_plugin:
|
||||
- 'mem0-plugin/.opencode-plugin/**'
|
||||
- 'integrations/mem0-plugin/.opencode-plugin/**'
|
||||
- '.github/workflows/opencode-plugin-checks.yml'
|
||||
- '.github/workflows/ci-gate.yml'
|
||||
pi_agent_plugin:
|
||||
- 'pi-agent-plugin/**'
|
||||
- 'integrations/pi-agent-plugin/**'
|
||||
- '.github/workflows/pi-agent-plugin-checks.yml'
|
||||
- '.github/workflows/ci-gate.yml'
|
||||
docs_llms_txt:
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
id-token: write
|
||||
defaults:
|
||||
run:
|
||||
working-directory: openclaw
|
||||
working-directory: integrations/openclaw
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
@@ -42,7 +42,7 @@ jobs:
|
||||
node-version: '22'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: openclaw/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/openclaw/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: pnpm install --frozen-lockfile
|
||||
|
||||
@@ -7,7 +7,7 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'openclaw/**'
|
||||
- 'integrations/openclaw/**'
|
||||
- '.github/workflows/openclaw-checks.yml'
|
||||
workflow_call:
|
||||
|
||||
@@ -27,13 +27,13 @@ jobs:
|
||||
with:
|
||||
node-version: 20
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: openclaw/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/openclaw/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: cd openclaw && pnpm install --frozen-lockfile
|
||||
run: cd integrations/openclaw && pnpm install --frozen-lockfile
|
||||
|
||||
- name: Type check
|
||||
run: cd openclaw && pnpm exec tsc --noEmit
|
||||
run: cd integrations/openclaw && pnpm exec tsc --noEmit
|
||||
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -53,20 +53,20 @@ jobs:
|
||||
with:
|
||||
node-version: ${{ matrix.node-version }}
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: openclaw/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/openclaw/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: cd openclaw && pnpm install --frozen-lockfile
|
||||
run: cd integrations/openclaw && pnpm install --frozen-lockfile
|
||||
|
||||
- name: Run tests with coverage
|
||||
run: cd openclaw && pnpm exec vitest run --coverage
|
||||
run: cd integrations/openclaw && pnpm exec vitest run --coverage
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
if: matrix.node-version == 20
|
||||
uses: codecov/codecov-action@v4
|
||||
with:
|
||||
flags: openclaw
|
||||
directory: openclaw/coverage
|
||||
directory: integrations/openclaw/coverage
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
|
||||
@@ -85,15 +85,15 @@ jobs:
|
||||
with:
|
||||
node-version: 20
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: openclaw/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/openclaw/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: cd openclaw && pnpm install --frozen-lockfile
|
||||
run: cd integrations/openclaw && pnpm install --frozen-lockfile
|
||||
|
||||
- name: Build
|
||||
run: cd openclaw && pnpm build
|
||||
run: cd integrations/openclaw && pnpm build
|
||||
|
||||
- name: Verify dist output exists
|
||||
run: |
|
||||
test -f openclaw/dist/index.js || (echo "Build output missing: dist/index.js" && exit 1)
|
||||
test -f openclaw/dist/index.d.ts || (echo "Build output missing: dist/index.d.ts" && exit 1)
|
||||
test -f integrations/openclaw/dist/index.js || (echo "Build output missing: dist/index.js" && exit 1)
|
||||
test -f integrations/openclaw/dist/index.d.ts || (echo "Build output missing: dist/index.d.ts" && exit 1)
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
id-token: write
|
||||
defaults:
|
||||
run:
|
||||
working-directory: mem0-plugin/.opencode-plugin
|
||||
working-directory: integrations/mem0-plugin/.opencode-plugin
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
|
||||
@@ -7,7 +7,7 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'mem0-plugin/.opencode-plugin/**'
|
||||
- 'integrations/mem0-plugin/.opencode-plugin/**'
|
||||
- '.github/workflows/opencode-plugin-checks.yml'
|
||||
workflow_call:
|
||||
|
||||
@@ -16,7 +16,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
working-directory: mem0-plugin/.opencode-plugin
|
||||
working-directory: integrations/mem0-plugin/.opencode-plugin
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
id-token: write
|
||||
defaults:
|
||||
run:
|
||||
working-directory: pi-agent-plugin
|
||||
working-directory: integrations/pi-agent-plugin
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
@@ -42,7 +42,7 @@ jobs:
|
||||
node-version: '22'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: pi-agent-plugin/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/pi-agent-plugin/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: pnpm install --frozen-lockfile
|
||||
|
||||
@@ -7,7 +7,7 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'pi-agent-plugin/**'
|
||||
- 'integrations/pi-agent-plugin/**'
|
||||
- '.github/workflows/pi-agent-plugin-checks.yml'
|
||||
workflow_call:
|
||||
|
||||
@@ -27,13 +27,13 @@ jobs:
|
||||
with:
|
||||
node-version: 20
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: pi-agent-plugin/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/pi-agent-plugin/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: cd pi-agent-plugin && pnpm install --frozen-lockfile
|
||||
run: cd integrations/pi-agent-plugin && pnpm install --frozen-lockfile
|
||||
|
||||
- name: Type check
|
||||
run: cd pi-agent-plugin && pnpm exec tsc --noEmit
|
||||
run: cd integrations/pi-agent-plugin && pnpm exec tsc --noEmit
|
||||
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -53,13 +53,13 @@ jobs:
|
||||
with:
|
||||
node-version: ${{ matrix.node-version }}
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: pi-agent-plugin/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/pi-agent-plugin/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: cd pi-agent-plugin && pnpm install --frozen-lockfile
|
||||
run: cd integrations/pi-agent-plugin && pnpm install --frozen-lockfile
|
||||
|
||||
- name: Run tests
|
||||
run: cd pi-agent-plugin && pnpm exec vitest run
|
||||
run: cd integrations/pi-agent-plugin && pnpm exec vitest run
|
||||
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -76,17 +76,17 @@ jobs:
|
||||
with:
|
||||
node-version: 20
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: pi-agent-plugin/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/pi-agent-plugin/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: cd pi-agent-plugin && pnpm install --frozen-lockfile
|
||||
run: cd integrations/pi-agent-plugin && pnpm install --frozen-lockfile
|
||||
|
||||
- name: Build
|
||||
run: cd pi-agent-plugin && pnpm build
|
||||
run: cd integrations/pi-agent-plugin && pnpm build
|
||||
|
||||
- name: Verify dist output exists
|
||||
run: |
|
||||
test -f pi-agent-plugin/dist/index.js || (echo "Build output missing: dist/index.js" && exit 1)
|
||||
test -f pi-agent-plugin/dist/index.d.ts || (echo "Build output missing: dist/index.d.ts" && exit 1)
|
||||
test -f pi-agent-plugin/dist/entry.js || (echo "Build output missing: dist/entry.js" && exit 1)
|
||||
test -f pi-agent-plugin/dist/entry.d.ts || (echo "Build output missing: dist/entry.d.ts" && exit 1)
|
||||
test -f integrations/pi-agent-plugin/dist/index.js || (echo "Build output missing: dist/index.js" && exit 1)
|
||||
test -f integrations/pi-agent-plugin/dist/index.d.ts || (echo "Build output missing: dist/index.d.ts" && exit 1)
|
||||
test -f integrations/pi-agent-plugin/dist/entry.js || (echo "Build output missing: dist/entry.js" && exit 1)
|
||||
test -f integrations/pi-agent-plugin/dist/entry.d.ts || (echo "Build output missing: dist/entry.d.ts" && exit 1)
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
id-token: write
|
||||
defaults:
|
||||
run:
|
||||
working-directory: vercel-ai-sdk
|
||||
working-directory: integrations/vercel-ai-sdk
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
@@ -42,7 +42,7 @@ jobs:
|
||||
node-version: '22'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
cache: 'pnpm'
|
||||
cache-dependency-path: vercel-ai-sdk/pnpm-lock.yaml
|
||||
cache-dependency-path: integrations/vercel-ai-sdk/pnpm-lock.yaml
|
||||
|
||||
- name: Install dependencies
|
||||
run: pnpm install --frozen-lockfile
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
[submodule "evaluation"]
|
||||
path = evaluation
|
||||
url = https://github.com/mem0ai/memory-benchmarks
|
||||
branch = main
|
||||
@@ -12,7 +12,7 @@ This file provides context for AI coding assistants (Claude Code, Cursor, GitHub
|
||||
|
||||
## Repository Structure
|
||||
|
||||
This is a **polyglot monorepo** containing Python and TypeScript packages, CLIs, servers, plugins, documentation, and evaluation tooling.
|
||||
This is a **polyglot monorepo** containing Python and TypeScript packages, CLIs, servers, plugins, and documentation.
|
||||
|
||||
### Key Directories
|
||||
|
||||
@@ -22,17 +22,18 @@ This is a **polyglot monorepo** containing Python and TypeScript packages, CLIs,
|
||||
| `mem0-ts/` | TypeScript SDK (`mem0ai` on npm) — client + OSS memory |
|
||||
| `cli/python/` | Python CLI (`mem0-cli` on PyPI) — Typer-based, entry point `mem0` |
|
||||
| `cli/node/` | Node CLI (`@mem0/cli` on npm) — Commander-based, entry point `mem0` |
|
||||
| `vercel-ai-sdk/` | `@mem0/vercel-ai-provider` — Vercel AI SDK memory provider |
|
||||
| `openclaw/` | `@mem0/openclaw-mem0` — OpenClaw plugin for Claude Code / AI editors |
|
||||
| `integrations/` | **Agent & editor integrations**, one directory per integration (see "Adding a New Integration") |
|
||||
| `integrations/mem0-plugin/` | AI editor plugins (Claude Code, Cursor, Codex) — MCP server connection, lifecycle hooks, skills. Contains nested `.opencode-plugin/` (`@mem0/opencode-plugin`) |
|
||||
| `integrations/openclaw/` | `@mem0/openclaw-mem0` — OpenClaw plugin for Claude Code / AI editors |
|
||||
| `integrations/pi-agent-plugin/` | `@mem0/pi-agent-plugin` — Pi Agent plugin |
|
||||
| `integrations/vercel-ai-sdk/` | `@mem0/vercel-ai-provider` — Vercel AI SDK memory provider |
|
||||
| `server/` | FastAPI REST server for self-hosted Mem0 (Docker: FastAPI + PostgreSQL/pgvector + Neo4j) |
|
||||
| `openmemory/` | Self-hosted memory platform — `api/` (FastAPI + Alembic + MCP server) and `ui/` (Next.js 15 + React 19) |
|
||||
| `mem0-plugin/` | AI editor plugins (Claude Code, Cursor, Codex) — MCP server connection, lifecycle hooks, skills |
|
||||
| `skills/` | Claude Code skill definitions. Reference skills (SDK knowledge, always-on): `mem0/`, `mem0-cli/`, `mem0-vercel-ai-sdk/`. Pipeline skills (run on demand): `mem0-integrate/`, `mem0-test-integration/` |
|
||||
| `skills/` | Claude Code skill definitions. Reference skills (SDK knowledge, always-on): `mem0/`, `mem0-cli/`, `mem0-vercel-ai-sdk/`. Pipeline skills (run on demand): `mem0-integrate/`, `mem0-test-integration/`, `mem0-oss-to-platform/` |
|
||||
| `docs/` | Documentation site (Mintlify) |
|
||||
| `tests/` | Python SDK tests (pytest) |
|
||||
| `evaluation/` | Benchmarking framework — LOCOMO evals, experiment runner, score generation |
|
||||
| `examples/` | Sample projects — demo apps, Chrome extension, multi-agent patterns |
|
||||
| `cookbooks/` | Jupyter notebooks — customer support chatbot, AutoGen integration |
|
||||
| `evaluation/` | Submodule → [`mem0ai/memory-benchmarks`](https://github.com/mem0ai/memory-benchmarks) — benchmarking (LOCOMO, LongMemEval, BEAM) lives in that repo |
|
||||
| `examples/` | Sample projects & runnable demos — apps, Chrome extension, multi-agent patterns, and Jupyter notebooks (`notebooks/`) |
|
||||
| `pr-reviews/` | Pull request review materials |
|
||||
| `scripts/` | Repo-wide utility scripts (e.g., `check-llms-txt-coverage.py` for docs/llms.txt sync) |
|
||||
|
||||
@@ -49,8 +50,8 @@ mem0 (Python SDK) mem0-ts (TypeScript SDK)
|
||||
|
||||
cli/python/ ──▶ mem0ai (optional, for OSS mode)
|
||||
cli/node/ ──▶ mem0ai (npm, for API calls)
|
||||
vercel-ai-sdk/ ──▶ ai, @ai-sdk/* providers
|
||||
openclaw/ ──▶ mem0ai (npm)
|
||||
integrations/vercel-ai-sdk/ ──▶ ai, @ai-sdk/* providers
|
||||
integrations/openclaw/ ──▶ mem0ai (npm)
|
||||
```
|
||||
|
||||
## Development Setup
|
||||
@@ -73,8 +74,8 @@ pre-commit install # install git hooks
|
||||
# TypeScript packages
|
||||
cd mem0-ts && pnpm install # TS SDK
|
||||
cd cli/node && pnpm install # Node CLI
|
||||
cd vercel-ai-sdk && pnpm install # Vercel AI provider
|
||||
cd openclaw && pnpm install # OpenClaw plugin
|
||||
cd integrations/vercel-ai-sdk && pnpm install # Vercel AI provider
|
||||
cd integrations/openclaw && pnpm install # OpenClaw plugin
|
||||
```
|
||||
|
||||
## Build, Lint, and Test Commands
|
||||
@@ -162,10 +163,10 @@ pnpm run dev # tsx src/index.ts (development)
|
||||
- **Test:** vitest (not jest)
|
||||
- **Framework:** Commander + Chalk + ora + cli-table3
|
||||
|
||||
### Vercel AI SDK Provider (`vercel-ai-sdk/`)
|
||||
### Vercel AI SDK Provider (`integrations/vercel-ai-sdk/`)
|
||||
|
||||
```bash
|
||||
cd vercel-ai-sdk
|
||||
cd integrations/vercel-ai-sdk
|
||||
pnpm install
|
||||
pnpm run build # tsup
|
||||
pnpm run lint # eslint
|
||||
@@ -180,10 +181,10 @@ pnpm run test:node # vitest (node runtime)
|
||||
- **Lint:** ESLint + Prettier
|
||||
- **Test:** jest + vitest (edge/node configs)
|
||||
|
||||
### OpenClaw Plugin (`openclaw/`)
|
||||
### OpenClaw Plugin (`integrations/openclaw/`)
|
||||
|
||||
```bash
|
||||
cd openclaw
|
||||
cd integrations/openclaw
|
||||
pnpm install
|
||||
pnpm run build # tsup
|
||||
pnpm run test # vitest run
|
||||
@@ -245,18 +246,19 @@ make docs # or: cd docs && mintlify dev
|
||||
- **API spec:** `docs/openapi.json`
|
||||
- **Structure:** `api-reference/`, `open-source/`, `platform/`, `integrations/`, `cookbooks/`, `core-concepts/`
|
||||
|
||||
### Evaluation (`evaluation/`)
|
||||
### Evaluation / Benchmarking
|
||||
|
||||
Benchmarking lives in the external [`mem0ai/memory-benchmarks`](https://github.com/mem0ai/memory-benchmarks) repo (LOCOMO + LongMemEval + BEAM). The in-repo `evaluation/` path is a **git submodule** pinned to that repo's `main` — populate it with `git submodule update --init evaluation` (or clone mem0 with `--recurse-submodules`), or clone the benchmarks repo standalone:
|
||||
|
||||
```bash
|
||||
cd evaluation
|
||||
make run-mem0-add # Run mem0 add experiments
|
||||
make run-mem0-search # Run mem0 search experiments
|
||||
make run-mem0-plus-add # With graph memory
|
||||
make run-mem0-plus-search # With graph memory
|
||||
make run-rag # RAG baseline
|
||||
make run-full-context # Full context baseline
|
||||
make run-langmem # LangMem comparison
|
||||
make run-openai # OpenAI comparison
|
||||
git clone https://github.com/mem0ai/memory-benchmarks.git
|
||||
cd memory-benchmarks
|
||||
pip install -r requirements.txt
|
||||
|
||||
# Run a benchmark (Mem0 Cloud; use docker compose for OSS)
|
||||
python -m benchmarks.locomo.run --project-name my-test --backend cloud --mem0-api-key $MEM0_API_KEY
|
||||
python -m benchmarks.longmemeval.run --project-name my-test --backend cloud --mem0-api-key $MEM0_API_KEY --all-questions
|
||||
python -m benchmarks.beam.run --project-name my-test --backend cloud --mem0-api-key $MEM0_API_KEY --chat-sizes 100K --conversations 0-9
|
||||
```
|
||||
|
||||
## Core APIs
|
||||
@@ -342,8 +344,8 @@ make run-openai # OpenAI comparison
|
||||
|---------|--------|-----------|---------------|
|
||||
| `mem0-ts/` | — | Prettier | jest |
|
||||
| `cli/node/` | Biome | Biome | vitest |
|
||||
| `vercel-ai-sdk/` | ESLint | Prettier | jest + vitest |
|
||||
| `openclaw/` | — | — | vitest |
|
||||
| `integrations/vercel-ai-sdk/` | ESLint | Prettier | jest + vitest |
|
||||
| `integrations/openclaw/` | — | — | vitest |
|
||||
|
||||
### Type Checking
|
||||
|
||||
@@ -381,14 +383,14 @@ Model Context Protocol support in multiple places:
|
||||
|
||||
- **Remote:** MCP server at `mcp.mem0.ai`
|
||||
- **Local:** MCP server in `openmemory/api/` (FastAPI-based)
|
||||
- **Plugin:** MCP tools in `mem0-plugin/` — 9 tools: `add_memory`, `search_memories`, `get_memories`, `get_memory`, `update_memory`, `delete_memory`, `delete_all_memories`, `delete_entities`, `list_entities`
|
||||
- **Plugin:** MCP tools in `integrations/mem0-plugin/` — 9 tools: `add_memory`, `search_memories`, `get_memories`, `get_memory`, `update_memory`, `delete_memory`, `delete_all_memories`, `delete_entities`, `list_entities`
|
||||
|
||||
### Plugin & Skills System
|
||||
|
||||
- `mem0-plugin/` provides integrations for Claude Code, Cursor, and Codex via MCP server connections and lifecycle hooks for automatic memory capture.
|
||||
- `integrations/mem0-plugin/` provides integrations for Claude Code, Cursor, and Codex via MCP server connections and lifecycle hooks for automatic memory capture.
|
||||
- `skills/` contains structured skill definitions for AI agents, split into two categories:
|
||||
- **Reference skills** (always-on SDK knowledge): `mem0` (Python + TS SDKs, framework integrations), `mem0-cli` (terminal workflows), `mem0-vercel-ai-sdk` (Vercel AI provider).
|
||||
- **Pipeline skills** (run on demand): `mem0-integrate` wires Mem0 into an existing repo via a TDD pipeline; `mem0-test-integration` verifies what the integrator produced on the same branch. The two are loosely coupled via `.mem0-integration/` artifacts.
|
||||
- **Pipeline skills** (run on demand): `mem0-integrate` wires Mem0 into an existing repo via a TDD pipeline; `mem0-test-integration` verifies what the integrator produced on the same branch (the two are loosely coupled via `.mem0-integration/` artifacts); `mem0-oss-to-platform` migrates an existing project from Mem0 OSS to the hosted Platform SDK (plan, then execute on approval).
|
||||
|
||||
### Adding a New Provider
|
||||
|
||||
@@ -402,6 +404,17 @@ To add a new LLM, embedding, vector store, or reranker provider:
|
||||
6. Add any new dependencies to the appropriate optional group in `pyproject.toml` (never to core `dependencies`)
|
||||
7. Follow the exact pattern of existing providers in the same category — match method signatures, error handling, and config structure
|
||||
|
||||
### Adding a New Integration
|
||||
|
||||
Agent/editor integrations live under `integrations/`. Each is a self-contained directory (its own `package.json`/lockfile, build, and tests). To add one:
|
||||
|
||||
1. Create `integrations/<name>/` and build the integration there.
|
||||
2. If it publishes to a registry, set `repository.directory: "integrations/<name>"` in its `package.json` so npm provenance links to the correct subdirectory.
|
||||
3. Add CI/CD under `.github/workflows/` (`<name>-checks.yml`, `<name>-cd.yml`). Use `integrations/<name>` in `paths:` triggers, `working-directory`, and `cache-dependency-path`. Register the release tag prefix in the `case` block in `release.yml` (keep the bare `v*` arm last). Keep workflow **filenames** stable — npm OIDC trusted publishing is pinned to repo + workflow filename.
|
||||
4. If it is a Claude Code / editor marketplace plugin, register its path in the five `marketplace.json` files (root + `.claude-plugin/`, `.cursor-plugin/`, `.codex-plugin/`, `.agents/plugins/`).
|
||||
5. Document it under `docs/integrations/` and add the page to `docs/docs.json` and `docs/llms.txt`.
|
||||
6. Add rows to the "Key Directories" table and the CI/CD tables in this file.
|
||||
|
||||
## CI/CD
|
||||
|
||||
### CI Workflows (automated testing)
|
||||
@@ -415,9 +428,9 @@ PR testing is orchestrated by a single entry point: **`ci-gate.yml` (CI Gate)**
|
||||
| TypeScript SDK | `ts-sdk-ci.yml` | Push to main (on `mem0-ts/`) | Prettier + build + jest on Node 20, 22 |
|
||||
| Python CLI | `cli-python-ci.yml` | Push to main (on `cli/python/`), manual | Ruff lint + pytest + hatch build on Python 3.10, 3.11, 3.12 |
|
||||
| Node CLI | `cli-node-ci.yml` | Push to main (on `cli/node/`), manual | Biome lint + tsc + vitest + tsup build on Node 20, 22 |
|
||||
| OpenClaw | `openclaw-checks.yml` | Push to main (on `openclaw/`), manual | tsc + vitest (with Codecov) + tsup build on Node 20, 22 |
|
||||
| OpenCode Plugin | `opencode-plugin-checks.yml` | Push to main (on `mem0-plugin/.opencode-plugin/`), manual | Bun: tsc type-check + build + dist artifact check |
|
||||
| Pi Agent Plugin | `pi-agent-plugin-checks.yml` | Push to main (on `pi-agent-plugin/`), manual | tsc + vitest + tsup build (dist artifact check) on Node 20, 22 |
|
||||
| OpenClaw | `openclaw-checks.yml` | Push to main (on `integrations/openclaw/`), manual | tsc + vitest (with Codecov) + tsup build on Node 20, 22 |
|
||||
| OpenCode Plugin | `opencode-plugin-checks.yml` | Push to main (on `integrations/mem0-plugin/.opencode-plugin/`), manual | Bun: tsc type-check + build + dist artifact check |
|
||||
| Pi Agent Plugin | `pi-agent-plugin-checks.yml` | Push to main (on `integrations/pi-agent-plugin/`), manual | tsc + vitest + tsup build (dist artifact check) on Node 20, 22 |
|
||||
| docs llms.txt | `docs-llms-txt-check.yml` | Manual | `docs/llms.txt` coverage check |
|
||||
|
||||
When adding a new package CI workflow: give it `workflow_call` (plus `push`/`workflow_dispatch` as needed, but no `pull_request` trigger), then register it in `ci-gate.yml` — a path filter under the `changes` job, a call job, and an entry in the gate job's `needs` list.
|
||||
|
||||
@@ -1026,7 +1026,8 @@ def get_user_preferences(user_id: str):
|
||||
### AutoGen Integration
|
||||
|
||||
```python
|
||||
from cookbooks.helper.mem0_teachability import Mem0Teachability
|
||||
# Mem0Teachability lives in examples/notebooks/helper/ — see examples/notebooks/mem0-autogen.ipynb
|
||||
from helper.mem0_teachability import Mem0Teachability
|
||||
from mem0 import Memory
|
||||
|
||||
# Add memory capability to AutoGen agents
|
||||
|
||||
@@ -186,9 +186,10 @@ npx skills add https://github.com/mem0ai/mem0 --skill mem0-vercel-ai-sdk
|
||||
```bash
|
||||
npx skills add https://github.com/mem0ai/mem0 --skill mem0-integrate
|
||||
npx skills add https://github.com/mem0ai/mem0 --skill mem0-test-integration
|
||||
npx skills add https://github.com/mem0ai/mem0 --skill mem0-oss-to-platform
|
||||
```
|
||||
|
||||
Use `/mem0-integrate` to wire Mem0 into an existing repo via a test-first pipeline, then `/mem0-test-integration` to verify. See the [skills catalog](./skills/) or [Vibecoding with Mem0](https://docs.mem0.ai/vibecoding) for the full picture.
|
||||
Use `/mem0-integrate` to wire Mem0 into an existing repo via a test-first pipeline, then `/mem0-test-integration` to verify. Use `/mem0-oss-to-platform` to migrate an existing project from Mem0 OSS to the hosted Platform SDK. See the [skills catalog](./skills/) or [Vibecoding with Mem0](https://docs.mem0.ai/vibecoding) for the full picture.
|
||||
|
||||
### Basic Usage
|
||||
|
||||
|
||||
Generated
+17
-16
@@ -9,6 +9,7 @@ overrides:
|
||||
langsmith@<0.6.0: ^0.6.0
|
||||
tar-fs@>=2.0.0 <2.1.4: ^2.1.4
|
||||
picomatch@<2.3.2: ^2.3.2
|
||||
postcss@<8.5.10: '>=8.5.10'
|
||||
|
||||
importers:
|
||||
|
||||
@@ -38,7 +39,7 @@ importers:
|
||||
version: 20.19.37
|
||||
tsup:
|
||||
specifier: ^8.0.0
|
||||
version: 8.5.1(postcss@8.5.8)(tsx@4.21.0)(typescript@5.9.3)
|
||||
version: 8.5.1(postcss@8.5.15)(tsx@4.21.0)(typescript@5.9.3)
|
||||
tsx:
|
||||
specifier: ^4.7.0
|
||||
version: 4.21.0
|
||||
@@ -829,8 +830,8 @@ packages:
|
||||
mz@2.7.0:
|
||||
resolution: {integrity: sha512-z81GNO7nnYMEhrGh9LeymoE4+Yr0Wn5McHIZMK5cfQCl+NDX08sCZgUc9/6MHni9IWuFLm1Z3HTCXu2z9fN62Q==}
|
||||
|
||||
nanoid@3.3.11:
|
||||
resolution: {integrity: sha512-N8SpfPUnUp1bK+PMYW8qSWdl9U+wwNWI4QKxOYDy9JAro3WMX7p2OeVRF9v+347pnakNevPmiHhNmZ2HbFA76w==}
|
||||
nanoid@3.3.12:
|
||||
resolution: {integrity: sha512-ZB9RH/39qpq5Vu6Y+NmUaFhQR6pp+M2Xt76XBnEwDaGcVAqhlvxrl3B2bKS5D3NH3QR76v3aSrKaF/Kiy7lEtQ==}
|
||||
engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1}
|
||||
hasBin: true
|
||||
|
||||
@@ -871,7 +872,7 @@ packages:
|
||||
engines: {node: '>= 18'}
|
||||
peerDependencies:
|
||||
jiti: '>=1.21.0'
|
||||
postcss: '>=8.0.9'
|
||||
postcss: '>=8.5.10'
|
||||
tsx: ^4.8.1
|
||||
yaml: ^2.4.2
|
||||
peerDependenciesMeta:
|
||||
@@ -884,8 +885,8 @@ packages:
|
||||
yaml:
|
||||
optional: true
|
||||
|
||||
postcss@8.5.8:
|
||||
resolution: {integrity: sha512-OW/rX8O/jXnm82Ey1k44pObPtdblfiuWnrd8X7GJ7emImCOstunGbXUpp7HdBrFQX6rJzn3sPT397Wp5aCwCHg==}
|
||||
postcss@8.5.15:
|
||||
resolution: {integrity: sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A==}
|
||||
engines: {node: ^10 || ^12 || >=14}
|
||||
|
||||
readdirp@4.1.2:
|
||||
@@ -997,7 +998,7 @@ packages:
|
||||
peerDependencies:
|
||||
'@microsoft/api-extractor': ^7.36.0
|
||||
'@swc/core': ^1
|
||||
postcss: ^8.4.12
|
||||
postcss: '>=8.5.10'
|
||||
typescript: '>=4.5.0'
|
||||
peerDependenciesMeta:
|
||||
'@microsoft/api-extractor':
|
||||
@@ -1671,7 +1672,7 @@ snapshots:
|
||||
object-assign: 4.1.1
|
||||
thenify-all: 1.6.0
|
||||
|
||||
nanoid@3.3.11: {}
|
||||
nanoid@3.3.12: {}
|
||||
|
||||
object-assign@4.1.1: {}
|
||||
|
||||
@@ -1707,16 +1708,16 @@ snapshots:
|
||||
mlly: 1.8.2
|
||||
pathe: 2.0.3
|
||||
|
||||
postcss-load-config@6.0.1(postcss@8.5.8)(tsx@4.21.0):
|
||||
postcss-load-config@6.0.1(postcss@8.5.15)(tsx@4.21.0):
|
||||
dependencies:
|
||||
lilconfig: 3.1.3
|
||||
optionalDependencies:
|
||||
postcss: 8.5.8
|
||||
postcss: 8.5.15
|
||||
tsx: 4.21.0
|
||||
|
||||
postcss@8.5.8:
|
||||
postcss@8.5.15:
|
||||
dependencies:
|
||||
nanoid: 3.3.11
|
||||
nanoid: 3.3.12
|
||||
picocolors: 1.1.1
|
||||
source-map-js: 1.2.1
|
||||
|
||||
@@ -1837,7 +1838,7 @@ snapshots:
|
||||
|
||||
ts-interface-checker@0.1.13: {}
|
||||
|
||||
tsup@8.5.1(postcss@8.5.8)(tsx@4.21.0)(typescript@5.9.3):
|
||||
tsup@8.5.1(postcss@8.5.15)(tsx@4.21.0)(typescript@5.9.3):
|
||||
dependencies:
|
||||
bundle-require: 5.1.0(esbuild@0.27.4)
|
||||
cac: 6.7.14
|
||||
@@ -1848,7 +1849,7 @@ snapshots:
|
||||
fix-dts-default-cjs-exports: 1.0.1
|
||||
joycon: 3.1.1
|
||||
picocolors: 1.1.1
|
||||
postcss-load-config: 6.0.1(postcss@8.5.8)(tsx@4.21.0)
|
||||
postcss-load-config: 6.0.1(postcss@8.5.15)(tsx@4.21.0)
|
||||
resolve-from: 5.0.0
|
||||
rollup: 4.60.0
|
||||
source-map: 0.7.6
|
||||
@@ -1857,7 +1858,7 @@ snapshots:
|
||||
tinyglobby: 0.2.15
|
||||
tree-kill: 1.2.2
|
||||
optionalDependencies:
|
||||
postcss: 8.5.8
|
||||
postcss: 8.5.15
|
||||
typescript: 5.9.3
|
||||
transitivePeerDependencies:
|
||||
- jiti
|
||||
@@ -1885,7 +1886,7 @@ snapshots:
|
||||
esbuild: 0.25.12
|
||||
fdir: 6.5.0(picomatch@4.0.4)
|
||||
picomatch: 4.0.4
|
||||
postcss: 8.5.8
|
||||
postcss: 8.5.15
|
||||
rollup: 4.60.0
|
||||
tinyglobby: 0.2.15
|
||||
optionalDependencies:
|
||||
|
||||
@@ -10,3 +10,4 @@ overrides:
|
||||
langsmith@<0.6.0: ^0.6.0
|
||||
tar-fs@>=2.0.0 <2.1.4: ^2.1.4
|
||||
picomatch@<2.3.2: ^2.3.2
|
||||
"postcss@<8.5.10": ">=8.5.10"
|
||||
|
||||
@@ -32,12 +32,13 @@ def _run(args: list[str], home_dir: str | None = None) -> subprocess.CompletedPr
|
||||
if key.startswith("MEM0_"):
|
||||
del env[key]
|
||||
env.pop("FORCE_COLOR", None)
|
||||
env["PYTHONIOENCODING"] = "utf-8"
|
||||
if home_dir:
|
||||
env["HOME"] = home_dir
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-m", "mem0_cli", *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
timeout=15,
|
||||
)
|
||||
@@ -99,12 +100,13 @@ class TestArgvPreprocessing:
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-m", "mem0_cli", "init", "--agent"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env={
|
||||
**{k: v for k, v in os.environ.items() if not k.startswith("MEM0_")},
|
||||
"HOME": clean_home,
|
||||
"MEM0_BASE_URL": "http://127.0.0.1:1", # blackhole
|
||||
"FORCE_COLOR": "0",
|
||||
"PYTHONIOENCODING": "utf-8",
|
||||
},
|
||||
timeout=15,
|
||||
)
|
||||
@@ -133,12 +135,13 @@ class TestJsonEnvelopeParity:
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-m", "mem0_cli", "init", "--agent", "--json"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env={
|
||||
**{k: v for k, v in os.environ.items() if not k.startswith("MEM0_")},
|
||||
"HOME": clean_home,
|
||||
"MEM0_BASE_URL": "http://127.0.0.1:1",
|
||||
"FORCE_COLOR": "0",
|
||||
"PYTHONIOENCODING": "utf-8",
|
||||
},
|
||||
timeout=15,
|
||||
)
|
||||
|
||||
@@ -49,6 +49,7 @@ def _run(
|
||||
if key.startswith("MEM0_"):
|
||||
del env[key]
|
||||
env.pop("FORCE_COLOR", None)
|
||||
env["PYTHONIOENCODING"] = "utf-8"
|
||||
if home_dir:
|
||||
env["HOME"] = home_dir
|
||||
if env_override:
|
||||
@@ -56,7 +57,7 @@ def _run(
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-m", "mem0_cli", *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
return subprocess.CompletedProcess(
|
||||
|
||||
@@ -67,7 +67,8 @@ class TestConfig:
|
||||
from mem0_cli.config import CONFIG_FILE
|
||||
|
||||
mode = os.stat(CONFIG_FILE).st_mode & 0o777
|
||||
assert mode == 0o600
|
||||
if os.name != "nt":
|
||||
assert mode == 0o600
|
||||
|
||||
def test_defaults_save_and_load(self, isolate_config):
|
||||
config = Mem0Config()
|
||||
|
||||
@@ -4,6 +4,39 @@ description: "Release notes for the OpenClaw plugin and agent harness."
|
||||
mode: "wide"
|
||||
---
|
||||
|
||||
<Update label="2026-06-12" description="v1.0.13">
|
||||
|
||||
**Fixes:**
|
||||
- **Custom categories payload:** `customCategories` (a `Record<string, string>` map) is now converted via the new `customCategoryMapToList()` helper into the `Array<Record<string, string>>` shape the Mem0 SDK expects on `add` calls — previously the raw object was passed as `custom_categories` and silently ignored ([#5345](https://github.com/mem0ai/mem0/pull/5345))
|
||||
- **Skip runtime setup during metadata registration:** `register()` now detects `registrationMode === "cli-metadata"`, registers only the CLI commands, and returns early — avoiding backend initialization, service/tool registration, and hook installation during OpenClaw's metadata-only registration pass ([#5383](https://github.com/mem0ai/mem0/pull/5383))
|
||||
|
||||
**Security:**
|
||||
- Bumped `mem0ai` from `3.0.3` to `3.0.7` (latest Node SDK) — includes the transitive axios CVE remediation shipped in `3.0.6` ([#5460](https://github.com/mem0ai/mem0/pull/5460))
|
||||
- Added pnpm override `uuid@<11.1.1` → `>=11.1.1` to resolve an open MEDIUM Dependabot alert ([#5489](https://github.com/mem0ai/mem0/pull/5489))
|
||||
|
||||
**Improvements:**
|
||||
- **Repo consolidation:** Plugin moved from repo-root `openclaw/` to `integrations/openclaw/`; `package.json` `repository.directory` updated to match so npm provenance links to the correct subdirectory ([#5491](https://github.com/mem0ai/mem0/pull/5491))
|
||||
|
||||
**Tests:**
|
||||
- Added `customCategoryMapToList` unit tests and a `PlatformProvider` test asserting `custom_categories` is passed to the Mem0 SDK as a list ([#5345](https://github.com/mem0ai/mem0/pull/5345))
|
||||
- Added a regression test asserting `cli-metadata` registration registers only CLI commands and triggers no runtime side effects ([#5383](https://github.com/mem0ai/mem0/pull/5383))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-06-02" description="v1.0.12">
|
||||
|
||||
**Docs:**
|
||||
- **Agent Mode onboarding:** README now documents an autonomous setup path for AI agents — `mem0 init --agent --json` mints an evaluation Mem0 API key with no email, OTP, or browser and exports it as `MEM0_API_KEY` for `openclaw mem0 init`; a human owner can later run `mem0 init --email <email>` to claim ownership without disrupting the agent ([#5123](https://github.com/mem0ai/mem0/pull/5123))
|
||||
|
||||
**Security:**
|
||||
- Added pnpm overrides to remediate advisories in transitive dependencies: `langsmith@<0.6.0` → `^0.6.0`, `picomatch@<2.3.2` → `^2.3.2`, `vite` → `^8.0.5`, and `@qdrant/js-client-rest` → `^1.18.0` ([#5294](https://github.com/mem0ai/mem0/pull/5294))
|
||||
|
||||
**Dependencies:**
|
||||
- Bumped `mem0ai` from `3.0.2` to `3.0.3` ([#5212](https://github.com/mem0ai/mem0/pull/5212))
|
||||
- Bumped dev dependencies `@vitest/coverage-v8` and `vitest` from `^4.0.18` to `^4.1.7`; added `vite@^8.0.5` and `@qdrant/js-client-rest@^1.18.0` ([#5294](https://github.com/mem0ai/mem0/pull/5294))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-04-29" description="v1.0.11">
|
||||
|
||||
**New Features:**
|
||||
|
||||
@@ -7,6 +7,22 @@ mode: "wide"
|
||||
<Tabs>
|
||||
<Tab title="Python">
|
||||
|
||||
<Update label="2026-06-13" description="v2.0.6">
|
||||
|
||||
**New Features:**
|
||||
- **Memory:** Add a contextual OSS-to-Platform notices system that surfaces occasional, situation-aware messages (first run, scale/performance thresholds, slow queries, and when temporal/decay features are relevant) pointing to the corresponding Mem0 Platform capabilities; disable via `MEM0_TELEMETRY=false` ([#5494](https://github.com/mem0ai/mem0/pull/5494))
|
||||
|
||||
**Bug Fixes:**
|
||||
- **Memory:** Prevent a crash in `parse_vision_messages` when vision support is disabled ([#5487](https://github.com/mem0ai/mem0/pull/5487))
|
||||
- **Vector Stores:** Expose the `https` option on the Qdrant vector store configuration so TLS endpoints can be targeted explicitly ([#5380](https://github.com/mem0ai/mem0/pull/5380))
|
||||
- **Vector Stores:** Use valid S3 Vectors entity index names, fixing index operations that failed on invalid names ([#5416](https://github.com/mem0ai/mem0/pull/5416))
|
||||
- **Vector Stores:** Fix `search()` crashing with a `TypeError` in the LangChain vector store when a result score is `None` ([#5072](https://github.com/mem0ai/mem0/pull/5072))
|
||||
- **Vector Stores:** Use `is not None` instead of a truthiness check for vector/payload in the PGVector `update()` path, so valid empty/zero values are no longer skipped ([#5488](https://github.com/mem0ai/mem0/pull/5488))
|
||||
- **Vector Stores:** Index the Valkey `memory` field as `TEXT` rather than `TAG` so full-text search behaves correctly ([#5443](https://github.com/mem0ai/mem0/pull/5443))
|
||||
- **Vector Stores:** Implement `$not` filter support in the ChromaDB vector store ([#5485](https://github.com/mem0ai/mem0/pull/5485))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-06-10" description="v2.0.5">
|
||||
|
||||
**New Features:**
|
||||
@@ -960,6 +976,17 @@ See the [OSS v1 to v2 migration guide](https://docs.mem0.ai/migration/oss-v1-to-
|
||||
|
||||
<Tab title="TypeScript">
|
||||
|
||||
<Update label="2026-06-13" description="v3.0.8">
|
||||
|
||||
**New Features:**
|
||||
- **Memory:** Add a contextual OSS-to-Platform notices system that surfaces occasional, situation-aware messages (first run, scale/performance thresholds, slow queries, and when temporal/decay features are relevant) pointing to the corresponding Mem0 Platform capabilities; disable via `MEM0_TELEMETRY=false` ([#5494](https://github.com/mem0ai/mem0/pull/5494))
|
||||
|
||||
**Security:**
|
||||
- **Dependencies:** Upgrade `@langchain/community` to `^1.1.18` to remediate CVE-2026-27795 and CVE-2026-26019 ([#5510](https://github.com/mem0ai/mem0/pull/5510))
|
||||
- **Dependencies:** Resolve all open MEDIUM Dependabot alerts via pnpm overrides ([#5489](https://github.com/mem0ai/mem0/pull/5489))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-06-10" description="v3.0.7">
|
||||
|
||||
**New Features:**
|
||||
|
||||
@@ -76,6 +76,7 @@ Let's see the available parameters for the `qdrant` config:
|
||||
| `path` | Path for the qdrant database | `/tmp/qdrant` |
|
||||
| `url` | Full URL for the qdrant server | `None` |
|
||||
| `api_key` | API key for the qdrant server | `None` |
|
||||
| `https` | Whether to force HTTPS on or off. `None` lets the client decide; set `False` for plain HTTP Qdrant with API key authentication. | `None` |
|
||||
| `on_disk` | For enabling persistent storage | `False` |
|
||||
</Tab>
|
||||
<Tab title="TypeScript">
|
||||
@@ -90,4 +91,4 @@ Let's see the available parameters for the `qdrant` config:
|
||||
| `apiKey` | API key for the Qdrant server | `None` |
|
||||
| `onDisk` | For enabling persistent storage | `False` |
|
||||
</Tab>
|
||||
</Tabs>
|
||||
</Tabs>
|
||||
|
||||
@@ -28,7 +28,7 @@ echo 'export MEM0_API_KEY="m0-your-api-key"' >> ~/.bashrc && source ~/.bashrc
|
||||
|
||||
```bash
|
||||
# Install the plugin (MCP server, hooks, scripts)
|
||||
npx degit mem0ai/mem0/mem0-plugin ~/.gemini/config/plugins/mem0
|
||||
npx degit mem0ai/mem0/integrations/mem0-plugin ~/.gemini/config/plugins/mem0
|
||||
```
|
||||
|
||||
This installs the MCP server, lifecycle hooks, and shared scripts.
|
||||
|
||||
@@ -43,7 +43,7 @@ bunx @mem0/opencode-plugin@latest install
|
||||
**Or let your agent do it** — paste this into OpenCode:
|
||||
|
||||
```
|
||||
Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/mem0-plugin/.opencode-plugin/README.md
|
||||
Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/integrations/mem0-plugin/.opencode-plugin/README.md
|
||||
```
|
||||
|
||||
All commands auto-add the plugin and MCP server to your `~/.config/opencode/opencode.json`. Restart OpenCode — you get the MCP server, lifecycle hooks, and all `/mem0:` slash commands.
|
||||
|
||||
@@ -59,6 +59,7 @@ For advanced settings, create `~/.pi/agent/mem0-config.json`:
|
||||
"userId": "your-username",
|
||||
"autoCapture": true,
|
||||
"defaultScope": "project",
|
||||
"searchThreshold": 0.3,
|
||||
"dream": {
|
||||
"enabled": true,
|
||||
"auto": true,
|
||||
@@ -75,12 +76,14 @@ For advanced settings, create `~/.pi/agent/mem0-config.json`:
|
||||
| `userId` | `string` | `$MEM0_USER_ID` or `"default"` | User identity for memory scoping |
|
||||
| `autoCapture` | `boolean` | `true` | Store facts from conversations automatically |
|
||||
| `defaultScope` | `string` | `"project"` | Default memory scope: `project`, `session`, or `global` |
|
||||
| `searchThreshold` | `number` | `0.3` | Minimum similarity score (0–1) a memory must reach to count as a match for `/mem0-search`, `/mem0-forget`, and `/mem0-pin`, enforced on each result's relevance score. Raise it to be stricter; lower it if relevant results are missed. |
|
||||
| `dream.enabled` | `boolean` | `true` | Enable dream consolidation |
|
||||
| `dream.auto` | `boolean` | `true` | Auto-trigger dreams when thresholds are met |
|
||||
| `dream.minHours` | `number` | `24` | Minimum hours between auto-dreams |
|
||||
| `dream.minSessions` | `number` | `5` | Minimum sessions before first auto-dream |
|
||||
| `dream.minMemories` | `number` | `20` | Minimum memories before auto-dream triggers |
|
||||
|
||||
|
||||
## What's Included
|
||||
|
||||
| Component | Description |
|
||||
|
||||
+2
-2
@@ -398,9 +398,9 @@ Each subdirectory is a Claude Code Skill (`SKILL.md` + supporting assets). Load
|
||||
|
||||
### Editor Plugin (shared glue)
|
||||
|
||||
Source: https://github.com/mem0ai/mem0/tree/main/mem0-plugin
|
||||
Source: https://github.com/mem0ai/mem0/tree/main/integrations/mem0-plugin
|
||||
|
||||
The `mem0-plugin/` directory provides MCP server connection, lifecycle hooks, and skill bundling for Claude Code, Cursor, Codex, OpenCode, and Antigravity. It exposes 9 MCP tools: `add_memory`, `search_memories`, `get_memories`, `get_memory`, `update_memory`, `delete_memory`, `delete_all_memories`, `delete_entities`, `list_entities`.
|
||||
The `integrations/mem0-plugin/` directory provides MCP server connection, lifecycle hooks, and skill bundling for Claude Code, Cursor, Codex, OpenCode, and Antigravity. It exposes 9 MCP tools: `add_memory`, `search_memories`, `get_memories`, `get_memory`, `update_memory`, `delete_memory`, `delete_all_memories`, `delete_entities`, `list_entities`.
|
||||
|
||||
Editor-specific setup docs (already listed above under `## Integrations > AI Coding Tools`):
|
||||
|
||||
|
||||
@@ -6,9 +6,7 @@ versionFrom: "Open Source"
|
||||
versionTo: "Platform"
|
||||
---
|
||||
|
||||
# Migrate from Open Source to Platform
|
||||
|
||||
Move your Mem0 implementation to managed infrastructure with enterprise features.
|
||||
## Overview
|
||||
|
||||
| Scope | Effort | Downtime |
|
||||
| --------------------- | -------------- | ---------------------------- |
|
||||
@@ -30,12 +28,29 @@ Move your Mem0 implementation to managed infrastructure with enterprise features
|
||||
- **Production Grade**: Auto-scaling, high availability, dedicated support
|
||||
</Info>
|
||||
|
||||
## Plan
|
||||
### Plan
|
||||
|
||||
1. **Sign up**: Create an account on <a href="https://app.mem0.ai?utm_source=oss&utm_medium=migration-oss-to-platform" rel="nofollow">Mem0 Platform</a>.
|
||||
2. **Get API Key**: Navigate to **Settings > API Keys** and generate a new key.
|
||||
3. **Review Usage**: Identify where you instantiate `Memory` and where you call `search` or `get_all`.
|
||||
|
||||
## Migrate with Agent Skill
|
||||
|
||||
Paste this prompt into your coding agent. It uses a migration skill to produce a plan; once you review and approve it, the agent implements the changes.
|
||||
|
||||
```text
|
||||
Migrate my project from Mem0 OSS to the Mem0 Platform SDK using the
|
||||
mem0-oss-to-platform skill in the mem0ai/mem0 repo, at
|
||||
skills/mem0-oss-to-platform/
|
||||
|
||||
Get the skill whichever way is easiest:
|
||||
- install it: npx skills add https://github.com/mem0ai/mem0 --skill mem0-oss-to-platform
|
||||
- if the mem0 repo is cloned locally, read it from skills/mem0-oss-to-platform/
|
||||
- otherwise fetch that folder from github.com/mem0ai/mem0 (SKILL.md + references/)
|
||||
|
||||
Then read SKILL.md and begin the migration.
|
||||
```
|
||||
|
||||
## Migrate
|
||||
|
||||
### 1. Import Memories Into Platform
|
||||
|
||||
@@ -45,10 +45,12 @@ Let your assistant execute an end-to-end workflow in an existing repo. Invoked a
|
||||
```bash
|
||||
npx skills add https://github.com/mem0ai/mem0 --skill mem0-integrate
|
||||
npx skills add https://github.com/mem0ai/mem0 --skill mem0-test-integration
|
||||
npx skills add https://github.com/mem0ai/mem0 --skill mem0-oss-to-platform
|
||||
```
|
||||
|
||||
- `/mem0-integrate` — wire Mem0 into an existing repository using a goal-driven, test-first pipeline. Detects the stack, asks whether to use Platform or OSS, writes failing tests first, and keeps the integration additive and feature-flagged.
|
||||
- `/mem0-test-integration` — verify what `/mem0-integrate` produced. Runs the repo's native test suite and a real end-to-end smoke flow against your API key, then produces a scorecard.
|
||||
- `/mem0-oss-to-platform` — migrate an existing project from Mem0 OSS to the hosted Platform SDK. Audits where Mem0 is used, writes a reviewable migration plan, then executes it on approval.
|
||||
|
||||
See the [skills index](https://github.com/mem0ai/mem0/tree/main/skills) for the full catalog.
|
||||
|
||||
|
||||
Submodule
+1
Submodule evaluation added at 4b61c5d31b
@@ -1,31 +0,0 @@
|
||||
|
||||
# Run the experiments
|
||||
run-mem0-add:
|
||||
python run_experiments.py --technique_type mem0 --method add
|
||||
|
||||
run-mem0-search:
|
||||
python run_experiments.py --technique_type mem0 --method search --output_folder results/ --top_k 30
|
||||
|
||||
run-mem0-plus-add:
|
||||
python run_experiments.py --technique_type mem0 --method add --is_graph
|
||||
|
||||
run-mem0-plus-search:
|
||||
python run_experiments.py --technique_type mem0 --method search --is_graph --output_folder results/ --top_k 30
|
||||
|
||||
run-rag:
|
||||
python run_experiments.py --technique_type rag --chunk_size 500 --num_chunks 1 --output_folder results/
|
||||
|
||||
run-full-context:
|
||||
python run_experiments.py --technique_type rag --chunk_size -1 --num_chunks 1 --output_folder results/
|
||||
|
||||
run-langmem:
|
||||
python run_experiments.py --technique_type langmem --output_folder results/
|
||||
|
||||
run-zep-add:
|
||||
python run_experiments.py --technique_type zep --method add --output_folder results/
|
||||
|
||||
run-zep-search:
|
||||
python run_experiments.py --technique_type zep --method search --output_folder results/
|
||||
|
||||
run-openai:
|
||||
python run_experiments.py --technique_type openai --output_folder results/
|
||||
@@ -1,198 +0,0 @@
|
||||
# Mem0: Building Production‑Ready AI Agents with Scalable Long‑Term Memory
|
||||
|
||||
[](https://arxiv.org/abs/2504.19413)
|
||||
[](https://mem0.ai/research)
|
||||
|
||||
This repository contains the code and dataset for our paper: **Mem0: Building Production‑Ready AI Agents with Scalable Long‑Term Memory**.
|
||||
|
||||
## 📋 Overview
|
||||
|
||||
This project evaluates Mem0 and compares it with different memory and retrieval techniques for AI systems:
|
||||
|
||||
1. **Established LOCOMO Benchmarks**: We evaluate against five established approaches from the literature: LoCoMo, ReadAgent, MemoryBank, MemGPT, and A-Mem.
|
||||
2. **Open-Source Memory Solutions**: We test promising open-source memory architectures including LangMem, which provides flexible memory management capabilities.
|
||||
3. **RAG Systems**: We implement Retrieval-Augmented Generation with various configurations, testing different chunk sizes and retrieval counts to optimize performance.
|
||||
4. **Full-Context Processing**: We examine the effectiveness of passing the entire conversation history within the context window of the LLM as a baseline approach.
|
||||
5. **Proprietary Memory Systems**: We evaluate OpenAI's built-in memory feature available in their ChatGPT interface to compare against commercial solutions.
|
||||
6. **Third-Party Memory Providers**: We incorporate Zep, a specialized memory management platform designed for AI agents, to assess the performance of dedicated memory infrastructure.
|
||||
|
||||
We test these techniques on the LOCOMO dataset, which contains conversational data with various question types to evaluate memory recall and understanding.
|
||||
|
||||
## 🔍 Dataset
|
||||
|
||||
The LOCOMO dataset used in our experiments can be downloaded from our Google Drive repository:
|
||||
|
||||
[Download LOCOMO Dataset](https://drive.google.com/drive/folders/1L-cTjTm0ohMsitsHg4dijSPJtqNflwX-?usp=drive_link)
|
||||
|
||||
The dataset contains conversational data specifically designed to test memory recall and understanding across various question types and complexity levels.
|
||||
|
||||
Place the dataset files in the `dataset/` directory:
|
||||
- `locomo10.json`: Original dataset
|
||||
- `locomo10_rag.json`: Dataset formatted for RAG experiments
|
||||
|
||||
## 📁 Project Structure
|
||||
|
||||
```
|
||||
.
|
||||
├── src/ # Source code for different memory techniques
|
||||
│ ├── mem0/ # Implementation of the Mem0 technique
|
||||
│ ├── openai/ # Implementation of the OpenAI memory
|
||||
│ ├── zep/ # Implementation of the Zep memory
|
||||
│ ├── rag.py # Implementation of the RAG technique
|
||||
│ └── langmem.py # Implementation of the Language-based memory
|
||||
├── metrics/ # Code for evaluation metrics
|
||||
├── results/ # Results of experiments
|
||||
├── dataset/ # Dataset files
|
||||
├── evals.py # Evaluation script
|
||||
├── run_experiments.py # Script to run experiments
|
||||
├── generate_scores.py # Script to generate scores from results
|
||||
└── prompts.py # Prompts used for the models
|
||||
```
|
||||
|
||||
## 🚀 Getting Started
|
||||
|
||||
### Prerequisites
|
||||
|
||||
Create a `.env` file with your API keys and configurations. The following keys are required:
|
||||
|
||||
```
|
||||
# OpenAI API key for GPT models and embeddings
|
||||
OPENAI_API_KEY="your-openai-api-key"
|
||||
|
||||
# Mem0 API keys (for Mem0 and Mem0+ techniques)
|
||||
MEM0_API_KEY="your-mem0-api-key"
|
||||
MEM0_PROJECT_ID="your-mem0-project-id"
|
||||
MEM0_ORGANIZATION_ID="your-mem0-organization-id"
|
||||
|
||||
# Model configuration
|
||||
MODEL="gpt-4o-mini" # or your preferred model
|
||||
EMBEDDING_MODEL="text-embedding-3-small" # or your preferred embedding model
|
||||
ZEP_API_KEY="api-key-from-zep"
|
||||
```
|
||||
|
||||
### Running Experiments
|
||||
|
||||
You can run experiments using the provided Makefile commands:
|
||||
|
||||
#### Memory Techniques
|
||||
|
||||
```bash
|
||||
# Run Mem0 experiments
|
||||
make run-mem0-add # Add memories using Mem0
|
||||
make run-mem0-search # Search memories using Mem0
|
||||
|
||||
# Run Mem0+ experiments (with graph-based search)
|
||||
make run-mem0-plus-add # Add memories using Mem0+
|
||||
make run-mem0-plus-search # Search memories using Mem0+
|
||||
|
||||
# Run RAG experiments
|
||||
make run-rag # Run RAG with chunk size 500
|
||||
make run-full-context # Run RAG with full context
|
||||
|
||||
# Run LangMem experiments
|
||||
make run-langmem # Run LangMem
|
||||
|
||||
# Run Zep experiments
|
||||
make run-zep-add # Add memories using Zep
|
||||
make run-zep-search # Search memories using Zep
|
||||
|
||||
# Run OpenAI experiments
|
||||
make run-openai # Run OpenAI experiments
|
||||
```
|
||||
|
||||
Alternatively, you can run experiments directly with custom parameters:
|
||||
|
||||
```bash
|
||||
python run_experiments.py --technique_type [mem0|rag|langmem] [additional parameters]
|
||||
```
|
||||
|
||||
#### Command-line Parameters:
|
||||
|
||||
| Parameter | Description | Default |
|
||||
|-----------|-------------|---------|
|
||||
| `--technique_type` | Memory technique to use (mem0, rag, langmem) | mem0 |
|
||||
| `--method` | Method to use (add, search) | add |
|
||||
| `--chunk_size` | Chunk size for processing | 1000 |
|
||||
| `--top_k` | Number of top memories to retrieve | 30 |
|
||||
| `--filter_memories` | Whether to filter memories | False |
|
||||
| `--is_graph` | Whether to use graph-based search | False |
|
||||
| `--num_chunks` | Number of chunks to process for RAG | 1 |
|
||||
|
||||
### 📊 Evaluation
|
||||
|
||||
To evaluate results, run:
|
||||
|
||||
```bash
|
||||
python evals.py --input_file [path_to_results] --output_file [output_path]
|
||||
```
|
||||
|
||||
This script:
|
||||
1. Processes each question-answer pair
|
||||
2. Calculates BLEU and F1 scores automatically
|
||||
3. Uses an LLM judge to evaluate answer correctness
|
||||
4. Saves the combined results to the output file
|
||||
|
||||
### 📈 Generating Scores
|
||||
|
||||
Generate final scores with:
|
||||
|
||||
```bash
|
||||
python generate_scores.py
|
||||
```
|
||||
|
||||
This script:
|
||||
1. Loads the evaluation metrics data
|
||||
2. Calculates mean scores for each category (BLEU, F1, LLM)
|
||||
3. Reports the number of questions per category
|
||||
4. Calculates overall mean scores across all categories
|
||||
|
||||
Example output:
|
||||
```
|
||||
Mean Scores Per Category:
|
||||
bleu_score f1_score llm_score count
|
||||
category
|
||||
1 0.xxxx 0.xxxx 0.xxxx xx
|
||||
2 0.xxxx 0.xxxx 0.xxxx xx
|
||||
3 0.xxxx 0.xxxx 0.xxxx xx
|
||||
|
||||
Overall Mean Scores:
|
||||
bleu_score 0.xxxx
|
||||
f1_score 0.xxxx
|
||||
llm_score 0.xxxx
|
||||
```
|
||||
|
||||
## 📏 Evaluation Metrics
|
||||
|
||||
We use several metrics to evaluate the performance of different memory techniques:
|
||||
|
||||
1. **BLEU Score**: Measures the similarity between the model's response and the ground truth
|
||||
2. **F1 Score**: Measures the harmonic mean of precision and recall
|
||||
3. **LLM Score**: A binary score (0 or 1) determined by an LLM judge evaluating the correctness of responses
|
||||
4. **Token Consumption**: Number of tokens required to generate final answer.
|
||||
5. **Latency**: Time required during search and to generate response.
|
||||
|
||||
## 📚 Citation
|
||||
|
||||
If you use this code or dataset in your research, please cite our paper:
|
||||
|
||||
```bibtex
|
||||
@article{mem0,
|
||||
title={Mem0: Building Production-Ready AI Agents with Scalable Long-Term Memory},
|
||||
author={Chhikara, Prateek and Khant, Dev and Aryan, Saket and Singh, Taranjeet and Yadav, Deshraj},
|
||||
journal={arXiv preprint arXiv:2504.19413},
|
||||
year={2025}
|
||||
}
|
||||
```
|
||||
|
||||
## 📄 License
|
||||
|
||||
[MIT License](LICENSE)
|
||||
|
||||
## 👥 Contributors
|
||||
|
||||
- [Prateek Chhikara](https://github.com/prateekchhikara)
|
||||
- [Dev Khant](https://github.com/Dev-Khant)
|
||||
- [Saket Aryan](https://github.com/whysosaket)
|
||||
- [Taranjeet Singh](https://github.com/taranjeet)
|
||||
- [Deshraj Yadav](https://github.com/deshraj)
|
||||
|
||||
@@ -1,81 +0,0 @@
|
||||
import argparse
|
||||
import concurrent.futures
|
||||
import json
|
||||
import threading
|
||||
from collections import defaultdict
|
||||
|
||||
from metrics.llm_judge import evaluate_llm_judge
|
||||
from metrics.utils import calculate_bleu_scores, calculate_metrics
|
||||
from tqdm import tqdm
|
||||
|
||||
|
||||
def process_item(item_data):
|
||||
k, v = item_data
|
||||
local_results = defaultdict(list)
|
||||
|
||||
for item in v:
|
||||
gt_answer = str(item["answer"])
|
||||
pred_answer = str(item["response"])
|
||||
category = str(item["category"])
|
||||
question = str(item["question"])
|
||||
|
||||
# Skip category 5
|
||||
if category == "5":
|
||||
continue
|
||||
|
||||
metrics = calculate_metrics(pred_answer, gt_answer)
|
||||
bleu_scores = calculate_bleu_scores(pred_answer, gt_answer)
|
||||
llm_score = evaluate_llm_judge(question, gt_answer, pred_answer)
|
||||
|
||||
local_results[k].append(
|
||||
{
|
||||
"question": question,
|
||||
"answer": gt_answer,
|
||||
"response": pred_answer,
|
||||
"category": category,
|
||||
"bleu_score": bleu_scores["bleu1"],
|
||||
"f1_score": metrics["f1"],
|
||||
"llm_score": llm_score,
|
||||
}
|
||||
)
|
||||
|
||||
return local_results
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Evaluate RAG results")
|
||||
parser.add_argument(
|
||||
"--input_file", type=str, default="results/rag_results_500_k1.json", help="Path to the input dataset file"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output_file", type=str, default="evaluation_metrics.json", help="Path to save the evaluation results"
|
||||
)
|
||||
parser.add_argument("--max_workers", type=int, default=10, help="Maximum number of worker threads")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
with open(args.input_file, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
results = defaultdict(list)
|
||||
results_lock = threading.Lock()
|
||||
|
||||
# Use ThreadPoolExecutor with specified workers
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=args.max_workers) as executor:
|
||||
futures = [executor.submit(process_item, item_data) for item_data in data.items()]
|
||||
|
||||
for future in tqdm(concurrent.futures.as_completed(futures), total=len(futures)):
|
||||
local_results = future.result()
|
||||
with results_lock:
|
||||
for k, items in local_results.items():
|
||||
results[k].extend(items)
|
||||
|
||||
# Save results to JSON file
|
||||
with open(args.output_file, "w") as f:
|
||||
json.dump(results, f, indent=4)
|
||||
|
||||
print(f"Results saved to {args.output_file}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,34 +0,0 @@
|
||||
import json
|
||||
|
||||
import pandas as pd
|
||||
|
||||
# Load the evaluation metrics data
|
||||
with open("evaluation_metrics.json", "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
# Flatten the data into a list of question items
|
||||
all_items = []
|
||||
for key in data:
|
||||
all_items.extend(data[key])
|
||||
|
||||
# Convert to DataFrame
|
||||
df = pd.DataFrame(all_items)
|
||||
|
||||
# Convert category to numeric type
|
||||
df["category"] = pd.to_numeric(df["category"])
|
||||
|
||||
# Calculate mean scores by category
|
||||
result = df.groupby("category").agg({"bleu_score": "mean", "f1_score": "mean", "llm_score": "mean"}).round(4)
|
||||
|
||||
# Add count of questions per category
|
||||
result["count"] = df.groupby("category").size()
|
||||
|
||||
# Print the results
|
||||
print("Mean Scores Per Category:")
|
||||
print(result)
|
||||
|
||||
# Calculate overall means
|
||||
overall_means = df.agg({"bleu_score": "mean", "f1_score": "mean", "llm_score": "mean"}).round(4)
|
||||
|
||||
print("\nOverall Mean Scores:")
|
||||
print(overall_means)
|
||||
@@ -1,130 +0,0 @@
|
||||
import argparse
|
||||
import json
|
||||
from collections import defaultdict
|
||||
|
||||
import numpy as np
|
||||
from openai import OpenAI
|
||||
|
||||
from mem0.memory.utils import extract_json
|
||||
|
||||
client = OpenAI()
|
||||
|
||||
ACCURACY_PROMPT = """
|
||||
Your task is to label an answer to a question as ’CORRECT’ or ’WRONG’. You will be given the following data:
|
||||
(1) a question (posed by one user to another user),
|
||||
(2) a ’gold’ (ground truth) answer,
|
||||
(3) a generated answer
|
||||
which you will score as CORRECT/WRONG.
|
||||
|
||||
The point of the question is to ask about something one user should know about the other user based on their prior conversations.
|
||||
The gold answer will usually be a concise and short answer that includes the referenced topic, for example:
|
||||
Question: Do you remember what I got the last time I went to Hawaii?
|
||||
Gold answer: A shell necklace
|
||||
The generated answer might be much longer, but you should be generous with your grading - as long as it touches on the same topic as the gold answer, it should be counted as CORRECT.
|
||||
|
||||
For time related questions, the gold answer will be a specific date, month, year, etc. The generated answer might be much longer or use relative time references (like "last Tuesday" or "next month"), but you should be generous with your grading - as long as it refers to the same date or time period as the gold answer, it should be counted as CORRECT. Even if the format differs (e.g., "May 7th" vs "7 May"), consider it CORRECT if it's the same date.
|
||||
|
||||
Now it's time for the real question:
|
||||
Question: {question}
|
||||
Gold answer: {gold_answer}
|
||||
Generated answer: {generated_answer}
|
||||
|
||||
First, provide a short (one sentence) explanation of your reasoning, then finish with CORRECT or WRONG.
|
||||
Do NOT include both CORRECT and WRONG in your response, or it will break the evaluation script.
|
||||
|
||||
Just return the label CORRECT or WRONG in a json format with the key as "label".
|
||||
"""
|
||||
|
||||
|
||||
def evaluate_llm_judge(question, gold_answer, generated_answer):
|
||||
"""Evaluate the generated answer against the gold answer using an LLM judge."""
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-4o-mini",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": ACCURACY_PROMPT.format(
|
||||
question=question, gold_answer=gold_answer, generated_answer=generated_answer
|
||||
),
|
||||
}
|
||||
],
|
||||
response_format={"type": "json_object"},
|
||||
temperature=0.0,
|
||||
)
|
||||
label = json.loads(extract_json(response.choices[0].message.content))["label"]
|
||||
return 1 if label == "CORRECT" else 0
|
||||
|
||||
|
||||
def main():
|
||||
"""Main function to evaluate RAG results using LLM judge."""
|
||||
parser = argparse.ArgumentParser(description="Evaluate RAG results using LLM judge")
|
||||
parser.add_argument(
|
||||
"--input_file",
|
||||
type=str,
|
||||
default="results/default_run_v4_k30_new_graph.json",
|
||||
help="Path to the input dataset file",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
dataset_path = args.input_file
|
||||
output_path = f"results/llm_judge_{dataset_path.split('/')[-1]}"
|
||||
|
||||
with open(dataset_path, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
LLM_JUDGE = defaultdict(list)
|
||||
RESULTS = defaultdict(list)
|
||||
|
||||
index = 0
|
||||
for k, v in data.items():
|
||||
for x in v:
|
||||
question = x["question"]
|
||||
gold_answer = x["answer"]
|
||||
generated_answer = x["response"]
|
||||
category = x["category"]
|
||||
|
||||
# Skip category 5
|
||||
if int(category) == 5:
|
||||
continue
|
||||
|
||||
# Evaluate the answer
|
||||
label = evaluate_llm_judge(question, gold_answer, generated_answer)
|
||||
LLM_JUDGE[category].append(label)
|
||||
|
||||
# Store the results
|
||||
RESULTS[index].append(
|
||||
{
|
||||
"question": question,
|
||||
"gt_answer": gold_answer,
|
||||
"response": generated_answer,
|
||||
"category": category,
|
||||
"llm_label": label,
|
||||
}
|
||||
)
|
||||
|
||||
# Save intermediate results
|
||||
with open(output_path, "w") as f:
|
||||
json.dump(RESULTS, f, indent=4)
|
||||
|
||||
# Print current accuracy for all categories
|
||||
print("All categories accuracy:")
|
||||
for cat, results in LLM_JUDGE.items():
|
||||
if results: # Only print if there are results for this category
|
||||
print(f" Category {cat}: {np.mean(results):.4f} ({sum(results)}/{len(results)})")
|
||||
print("------------------------------------------")
|
||||
index += 1
|
||||
|
||||
# Save final results
|
||||
with open(output_path, "w") as f:
|
||||
json.dump(RESULTS, f, indent=4)
|
||||
|
||||
# Print final summary
|
||||
print("PATH: ", dataset_path)
|
||||
print("------------------------------------------")
|
||||
for k, v in LLM_JUDGE.items():
|
||||
print(k, np.mean(v))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,211 +0,0 @@
|
||||
"""
|
||||
Borrowed from https://github.com/WujiangXu/AgenticMemory/blob/main/utils.py
|
||||
|
||||
@article{xu2025mem,
|
||||
title={A-mem: Agentic memory for llm agents},
|
||||
author={Xu, Wujiang and Liang, Zujie and Mei, Kai and Gao, Hang and Tan, Juntao
|
||||
and Zhang, Yongfeng},
|
||||
journal={arXiv preprint arXiv:2502.12110},
|
||||
year={2025}
|
||||
}
|
||||
"""
|
||||
|
||||
import statistics
|
||||
from collections import defaultdict
|
||||
from typing import Dict, List, Union
|
||||
|
||||
import nltk
|
||||
from bert_score import score as bert_score
|
||||
from nltk.translate.bleu_score import SmoothingFunction, sentence_bleu
|
||||
from nltk.translate.meteor_score import meteor_score
|
||||
from rouge_score import rouge_scorer
|
||||
from sentence_transformers import SentenceTransformer
|
||||
|
||||
# from load_dataset import load_locomo_dataset, QA, Turn, Session, Conversation
|
||||
from sentence_transformers.util import pytorch_cos_sim
|
||||
|
||||
# Download required NLTK data
|
||||
try:
|
||||
nltk.download("punkt", quiet=True)
|
||||
nltk.download("wordnet", quiet=True)
|
||||
except Exception as e:
|
||||
print(f"Error downloading NLTK data: {e}")
|
||||
|
||||
# Initialize SentenceTransformer model (this will be reused)
|
||||
try:
|
||||
sentence_model = SentenceTransformer("all-MiniLM-L6-v2")
|
||||
except Exception as e:
|
||||
print(f"Warning: Could not load SentenceTransformer model: {e}")
|
||||
sentence_model = None
|
||||
|
||||
|
||||
def simple_tokenize(text):
|
||||
"""Simple tokenization function."""
|
||||
# Convert to string if not already
|
||||
text = str(text)
|
||||
return text.lower().replace(".", " ").replace(",", " ").replace("!", " ").replace("?", " ").split()
|
||||
|
||||
|
||||
def calculate_rouge_scores(prediction: str, reference: str) -> Dict[str, float]:
|
||||
"""Calculate ROUGE scores for prediction against reference."""
|
||||
scorer = rouge_scorer.RougeScorer(["rouge1", "rouge2", "rougeL"], use_stemmer=True)
|
||||
scores = scorer.score(reference, prediction)
|
||||
return {
|
||||
"rouge1_f": scores["rouge1"].fmeasure,
|
||||
"rouge2_f": scores["rouge2"].fmeasure,
|
||||
"rougeL_f": scores["rougeL"].fmeasure,
|
||||
}
|
||||
|
||||
|
||||
def calculate_bleu_scores(prediction: str, reference: str) -> Dict[str, float]:
|
||||
"""Calculate BLEU scores with different n-gram settings."""
|
||||
pred_tokens = nltk.word_tokenize(prediction.lower())
|
||||
ref_tokens = [nltk.word_tokenize(reference.lower())]
|
||||
|
||||
weights_list = [(1, 0, 0, 0), (0.5, 0.5, 0, 0), (0.33, 0.33, 0.33, 0), (0.25, 0.25, 0.25, 0.25)]
|
||||
smooth = SmoothingFunction().method1
|
||||
|
||||
scores = {}
|
||||
for n, weights in enumerate(weights_list, start=1):
|
||||
try:
|
||||
score = sentence_bleu(ref_tokens, pred_tokens, weights=weights, smoothing_function=smooth)
|
||||
except Exception as e:
|
||||
print(f"Error calculating BLEU score: {e}")
|
||||
score = 0.0
|
||||
scores[f"bleu{n}"] = score
|
||||
|
||||
return scores
|
||||
|
||||
|
||||
def calculate_bert_scores(prediction: str, reference: str) -> Dict[str, float]:
|
||||
"""Calculate BERTScore for semantic similarity."""
|
||||
try:
|
||||
P, R, F1 = bert_score([prediction], [reference], lang="en", verbose=False)
|
||||
return {"bert_precision": P.item(), "bert_recall": R.item(), "bert_f1": F1.item()}
|
||||
except Exception as e:
|
||||
print(f"Error calculating BERTScore: {e}")
|
||||
return {"bert_precision": 0.0, "bert_recall": 0.0, "bert_f1": 0.0}
|
||||
|
||||
|
||||
def calculate_meteor_score(prediction: str, reference: str) -> float:
|
||||
"""Calculate METEOR score for the prediction."""
|
||||
try:
|
||||
return meteor_score([reference.split()], prediction.split())
|
||||
except Exception as e:
|
||||
print(f"Error calculating METEOR score: {e}")
|
||||
return 0.0
|
||||
|
||||
|
||||
def calculate_sentence_similarity(prediction: str, reference: str) -> float:
|
||||
"""Calculate sentence embedding similarity using SentenceBERT."""
|
||||
if sentence_model is None:
|
||||
return 0.0
|
||||
try:
|
||||
# Encode sentences
|
||||
embedding1 = sentence_model.encode([prediction], convert_to_tensor=True)
|
||||
embedding2 = sentence_model.encode([reference], convert_to_tensor=True)
|
||||
|
||||
# Calculate cosine similarity
|
||||
similarity = pytorch_cos_sim(embedding1, embedding2).item()
|
||||
return float(similarity)
|
||||
except Exception as e:
|
||||
print(f"Error calculating sentence similarity: {e}")
|
||||
return 0.0
|
||||
|
||||
|
||||
def calculate_metrics(prediction: str, reference: str) -> Dict[str, float]:
|
||||
"""Calculate comprehensive evaluation metrics for a prediction."""
|
||||
# Handle empty or None values
|
||||
if not prediction or not reference:
|
||||
return {
|
||||
"exact_match": 0,
|
||||
"f1": 0.0,
|
||||
"rouge1_f": 0.0,
|
||||
"rouge2_f": 0.0,
|
||||
"rougeL_f": 0.0,
|
||||
"bleu1": 0.0,
|
||||
"bleu2": 0.0,
|
||||
"bleu3": 0.0,
|
||||
"bleu4": 0.0,
|
||||
"bert_f1": 0.0,
|
||||
"meteor": 0.0,
|
||||
"sbert_similarity": 0.0,
|
||||
}
|
||||
|
||||
# Convert to strings if they're not already
|
||||
prediction = str(prediction).strip()
|
||||
reference = str(reference).strip()
|
||||
|
||||
# Calculate exact match
|
||||
exact_match = int(prediction.lower() == reference.lower())
|
||||
|
||||
# Calculate token-based F1 score
|
||||
pred_tokens = set(simple_tokenize(prediction))
|
||||
ref_tokens = set(simple_tokenize(reference))
|
||||
common_tokens = pred_tokens & ref_tokens
|
||||
|
||||
if not pred_tokens or not ref_tokens:
|
||||
f1 = 0.0
|
||||
else:
|
||||
precision = len(common_tokens) / len(pred_tokens)
|
||||
recall = len(common_tokens) / len(ref_tokens)
|
||||
f1 = 2 * precision * recall / (precision + recall) if (precision + recall) > 0 else 0.0
|
||||
|
||||
# Calculate all scores
|
||||
bleu_scores = calculate_bleu_scores(prediction, reference)
|
||||
|
||||
# Combine all metrics
|
||||
metrics = {
|
||||
"exact_match": exact_match,
|
||||
"f1": f1,
|
||||
**bleu_scores,
|
||||
}
|
||||
|
||||
return metrics
|
||||
|
||||
|
||||
def aggregate_metrics(
|
||||
all_metrics: List[Dict[str, float]], all_categories: List[int]
|
||||
) -> Dict[str, Dict[str, Union[float, Dict[str, float]]]]:
|
||||
"""Calculate aggregate statistics for all metrics, split by category."""
|
||||
if not all_metrics:
|
||||
return {}
|
||||
|
||||
# Initialize aggregates for overall and per-category metrics
|
||||
aggregates = defaultdict(list)
|
||||
category_aggregates = defaultdict(lambda: defaultdict(list))
|
||||
|
||||
# Collect all values for each metric, both overall and per category
|
||||
for metrics, category in zip(all_metrics, all_categories):
|
||||
for metric_name, value in metrics.items():
|
||||
aggregates[metric_name].append(value)
|
||||
category_aggregates[category][metric_name].append(value)
|
||||
|
||||
# Calculate statistics for overall metrics
|
||||
results = {"overall": {}}
|
||||
|
||||
for metric_name, values in aggregates.items():
|
||||
results["overall"][metric_name] = {
|
||||
"mean": statistics.mean(values),
|
||||
"std": statistics.stdev(values) if len(values) > 1 else 0.0,
|
||||
"median": statistics.median(values),
|
||||
"min": min(values),
|
||||
"max": max(values),
|
||||
"count": len(values),
|
||||
}
|
||||
|
||||
# Calculate statistics for each category
|
||||
for category in sorted(category_aggregates.keys()):
|
||||
results[f"category_{category}"] = {}
|
||||
for metric_name, values in category_aggregates[category].items():
|
||||
if values: # Only calculate if we have values for this category
|
||||
results[f"category_{category}"][metric_name] = {
|
||||
"mean": statistics.mean(values),
|
||||
"std": statistics.stdev(values) if len(values) > 1 else 0.0,
|
||||
"median": statistics.median(values),
|
||||
"min": min(values),
|
||||
"max": max(values),
|
||||
"count": len(values),
|
||||
}
|
||||
|
||||
return results
|
||||
@@ -1,147 +0,0 @@
|
||||
ANSWER_PROMPT_GRAPH = """
|
||||
You are an intelligent memory assistant tasked with retrieving accurate information from
|
||||
conversation memories.
|
||||
|
||||
# CONTEXT:
|
||||
You have access to memories from two speakers in a conversation. These memories contain
|
||||
timestamped information that may be relevant to answering the question. You also have
|
||||
access to knowledge graph relations for each user, showing connections between entities,
|
||||
concepts, and events relevant to that user.
|
||||
|
||||
# INSTRUCTIONS:
|
||||
1. Carefully analyze all provided memories from both speakers
|
||||
2. Pay special attention to the timestamps to determine the answer
|
||||
3. If the question asks about a specific event or fact, look for direct evidence in the
|
||||
memories
|
||||
4. If the memories contain contradictory information, prioritize the most recent memory
|
||||
5. If there is a question about time references (like "last year", "two months ago",
|
||||
etc.), calculate the actual date based on the memory timestamp. For example, if a
|
||||
memory from 4 May 2022 mentions "went to India last year," then the trip occurred
|
||||
in 2021.
|
||||
6. Always convert relative time references to specific dates, months, or years. For
|
||||
example, convert "last year" to "2022" or "two months ago" to "March 2023" based
|
||||
on the memory timestamp. Ignore the reference while answering the question.
|
||||
7. Focus only on the content of the memories from both speakers. Do not confuse
|
||||
character names mentioned in memories with the actual users who created those
|
||||
memories.
|
||||
8. The answer should be less than 5-6 words.
|
||||
9. Use the knowledge graph relations to understand the user's knowledge network and
|
||||
identify important relationships between entities in the user's world.
|
||||
|
||||
# APPROACH (Think step by step):
|
||||
1. First, examine all memories that contain information related to the question
|
||||
2. Examine the timestamps and content of these memories carefully
|
||||
3. Look for explicit mentions of dates, times, locations, or events that answer the
|
||||
question
|
||||
4. If the answer requires calculation (e.g., converting relative time references),
|
||||
show your work
|
||||
5. Analyze the knowledge graph relations to understand the user's knowledge context
|
||||
6. Formulate a precise, concise answer based solely on the evidence in the memories
|
||||
7. Double-check that your answer directly addresses the question asked
|
||||
8. Ensure your final answer is specific and avoids vague time references
|
||||
|
||||
Memories for user {{speaker_1_user_id}}:
|
||||
|
||||
{{speaker_1_memories}}
|
||||
|
||||
Relations for user {{speaker_1_user_id}}:
|
||||
|
||||
{{speaker_1_graph_memories}}
|
||||
|
||||
Memories for user {{speaker_2_user_id}}:
|
||||
|
||||
{{speaker_2_memories}}
|
||||
|
||||
Relations for user {{speaker_2_user_id}}:
|
||||
|
||||
{{speaker_2_graph_memories}}
|
||||
|
||||
Question: {{question}}
|
||||
|
||||
Answer:
|
||||
"""
|
||||
|
||||
|
||||
ANSWER_PROMPT = """
|
||||
You are an intelligent memory assistant tasked with retrieving accurate information from conversation memories.
|
||||
|
||||
# CONTEXT:
|
||||
You have access to memories from two speakers in a conversation. These memories contain
|
||||
timestamped information that may be relevant to answering the question.
|
||||
|
||||
# INSTRUCTIONS:
|
||||
1. Carefully analyze all provided memories from both speakers
|
||||
2. Pay special attention to the timestamps to determine the answer
|
||||
3. If the question asks about a specific event or fact, look for direct evidence in the memories
|
||||
4. If the memories contain contradictory information, prioritize the most recent memory
|
||||
5. If there is a question about time references (like "last year", "two months ago", etc.),
|
||||
calculate the actual date based on the memory timestamp. For example, if a memory from
|
||||
4 May 2022 mentions "went to India last year," then the trip occurred in 2021.
|
||||
6. Always convert relative time references to specific dates, months, or years. For example,
|
||||
convert "last year" to "2022" or "two months ago" to "March 2023" based on the memory
|
||||
timestamp. Ignore the reference while answering the question.
|
||||
7. Focus only on the content of the memories from both speakers. Do not confuse character
|
||||
names mentioned in memories with the actual users who created those memories.
|
||||
8. The answer should be less than 5-6 words.
|
||||
|
||||
# APPROACH (Think step by step):
|
||||
1. First, examine all memories that contain information related to the question
|
||||
2. Examine the timestamps and content of these memories carefully
|
||||
3. Look for explicit mentions of dates, times, locations, or events that answer the question
|
||||
4. If the answer requires calculation (e.g., converting relative time references), show your work
|
||||
5. Formulate a precise, concise answer based solely on the evidence in the memories
|
||||
6. Double-check that your answer directly addresses the question asked
|
||||
7. Ensure your final answer is specific and avoids vague time references
|
||||
|
||||
Memories for user {{speaker_1_user_id}}:
|
||||
|
||||
{{speaker_1_memories}}
|
||||
|
||||
Memories for user {{speaker_2_user_id}}:
|
||||
|
||||
{{speaker_2_memories}}
|
||||
|
||||
Question: {{question}}
|
||||
|
||||
Answer:
|
||||
"""
|
||||
|
||||
|
||||
ANSWER_PROMPT_ZEP = """
|
||||
You are an intelligent memory assistant tasked with retrieving accurate information from conversation memories.
|
||||
|
||||
# CONTEXT:
|
||||
You have access to memories from a conversation. These memories contain
|
||||
timestamped information that may be relevant to answering the question.
|
||||
|
||||
# INSTRUCTIONS:
|
||||
1. Carefully analyze all provided memories
|
||||
2. Pay special attention to the timestamps to determine the answer
|
||||
3. If the question asks about a specific event or fact, look for direct evidence in the memories
|
||||
4. If the memories contain contradictory information, prioritize the most recent memory
|
||||
5. If there is a question about time references (like "last year", "two months ago", etc.),
|
||||
calculate the actual date based on the memory timestamp. For example, if a memory from
|
||||
4 May 2022 mentions "went to India last year," then the trip occurred in 2021.
|
||||
6. Always convert relative time references to specific dates, months, or years. For example,
|
||||
convert "last year" to "2022" or "two months ago" to "March 2023" based on the memory
|
||||
timestamp. Ignore the reference while answering the question.
|
||||
7. Focus only on the content of the memories. Do not confuse character
|
||||
names mentioned in memories with the actual users who created those memories.
|
||||
8. The answer should be less than 5-6 words.
|
||||
|
||||
# APPROACH (Think step by step):
|
||||
1. First, examine all memories that contain information related to the question
|
||||
2. Examine the timestamps and content of these memories carefully
|
||||
3. Look for explicit mentions of dates, times, locations, or events that answer the question
|
||||
4. If the answer requires calculation (e.g., converting relative time references), show your work
|
||||
5. Formulate a precise, concise answer based solely on the evidence in the memories
|
||||
6. Double-check that your answer directly addresses the question asked
|
||||
7. Ensure your final answer is specific and avoids vague time references
|
||||
|
||||
Memories:
|
||||
|
||||
{{memories}}
|
||||
|
||||
Question: {{question}}
|
||||
Answer:
|
||||
"""
|
||||
@@ -1,75 +0,0 @@
|
||||
import argparse
|
||||
import os
|
||||
|
||||
from src.langmem import LangMemManager
|
||||
from src.memzero.add import MemoryADD
|
||||
from src.memzero.search import MemorySearch
|
||||
from src.openai.predict import OpenAIPredict
|
||||
from src.rag import RAGManager
|
||||
from src.utils import METHODS, TECHNIQUES
|
||||
from src.zep.add import ZepAdd
|
||||
from src.zep.search import ZepSearch
|
||||
|
||||
|
||||
class Experiment:
|
||||
def __init__(self, technique_type, chunk_size):
|
||||
self.technique_type = technique_type
|
||||
self.chunk_size = chunk_size
|
||||
|
||||
def run(self):
|
||||
print(f"Running experiment with technique: {self.technique_type}, chunk size: {self.chunk_size}")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Run memory experiments")
|
||||
parser.add_argument("--technique_type", choices=TECHNIQUES, default="mem0", help="Memory technique to use")
|
||||
parser.add_argument("--method", choices=METHODS, default="add", help="Method to use")
|
||||
parser.add_argument("--chunk_size", type=int, default=1000, help="Chunk size for processing")
|
||||
parser.add_argument("--output_folder", type=str, default="results/", help="Output path for results")
|
||||
parser.add_argument("--top_k", type=int, default=30, help="Number of top memories to retrieve")
|
||||
parser.add_argument("--filter_memories", action="store_true", default=False, help="Whether to filter memories")
|
||||
parser.add_argument("--is_graph", action="store_true", default=False, help="Whether to use graph-based search")
|
||||
parser.add_argument("--num_chunks", type=int, default=1, help="Number of chunks to process")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Add your experiment logic here
|
||||
print(f"Running experiments with technique: {args.technique_type}, chunk size: {args.chunk_size}")
|
||||
|
||||
if args.technique_type == "mem0":
|
||||
if args.method == "add":
|
||||
memory_manager = MemoryADD(data_path="dataset/locomo10.json", is_graph=args.is_graph)
|
||||
memory_manager.process_all_conversations()
|
||||
elif args.method == "search":
|
||||
output_file_path = os.path.join(
|
||||
args.output_folder,
|
||||
f"mem0_results_top_{args.top_k}_filter_{args.filter_memories}_graph_{args.is_graph}.json",
|
||||
)
|
||||
memory_searcher = MemorySearch(output_file_path, args.top_k, args.filter_memories, args.is_graph)
|
||||
memory_searcher.process_data_file("dataset/locomo10.json")
|
||||
elif args.technique_type == "rag":
|
||||
output_file_path = os.path.join(args.output_folder, f"rag_results_{args.chunk_size}_k{args.num_chunks}.json")
|
||||
rag_manager = RAGManager(data_path="dataset/locomo10_rag.json", chunk_size=args.chunk_size, k=args.num_chunks)
|
||||
rag_manager.process_all_conversations(output_file_path)
|
||||
elif args.technique_type == "langmem":
|
||||
output_file_path = os.path.join(args.output_folder, "langmem_results.json")
|
||||
langmem_manager = LangMemManager(dataset_path="dataset/locomo10_rag.json")
|
||||
langmem_manager.process_all_conversations(output_file_path)
|
||||
elif args.technique_type == "zep":
|
||||
if args.method == "add":
|
||||
zep_manager = ZepAdd(data_path="dataset/locomo10.json")
|
||||
zep_manager.process_all_conversations("1")
|
||||
elif args.method == "search":
|
||||
output_file_path = os.path.join(args.output_folder, "zep_search_results.json")
|
||||
zep_manager = ZepSearch()
|
||||
zep_manager.process_data_file("dataset/locomo10.json", "1", output_file_path)
|
||||
elif args.technique_type == "openai":
|
||||
output_file_path = os.path.join(args.output_folder, "openai_results.json")
|
||||
openai_manager = OpenAIPredict()
|
||||
openai_manager.process_data_file("dataset/locomo10.json", output_file_path)
|
||||
else:
|
||||
raise ValueError(f"Invalid technique type: {args.technique_type}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,185 +0,0 @@
|
||||
import json
|
||||
import multiprocessing as mp
|
||||
import os
|
||||
import time
|
||||
from collections import defaultdict
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from jinja2 import Template
|
||||
from langgraph.checkpoint.memory import MemorySaver
|
||||
from langgraph.prebuilt import create_react_agent
|
||||
from langgraph.store.memory import InMemoryStore
|
||||
from langgraph.utils.config import get_store
|
||||
from langmem import create_manage_memory_tool, create_search_memory_tool
|
||||
from openai import OpenAI
|
||||
from prompts import ANSWER_PROMPT
|
||||
from tqdm import tqdm
|
||||
|
||||
load_dotenv()
|
||||
|
||||
client = OpenAI()
|
||||
|
||||
ANSWER_PROMPT_TEMPLATE = Template(ANSWER_PROMPT)
|
||||
|
||||
|
||||
def get_answer(question, speaker_1_user_id, speaker_1_memories, speaker_2_user_id, speaker_2_memories):
|
||||
prompt = ANSWER_PROMPT_TEMPLATE.render(
|
||||
question=question,
|
||||
speaker_1_user_id=speaker_1_user_id,
|
||||
speaker_1_memories=speaker_1_memories,
|
||||
speaker_2_user_id=speaker_2_user_id,
|
||||
speaker_2_memories=speaker_2_memories,
|
||||
)
|
||||
|
||||
t1 = time.time()
|
||||
response = client.chat.completions.create(
|
||||
model=os.getenv("MODEL"), messages=[{"role": "system", "content": prompt}], temperature=0.0
|
||||
)
|
||||
t2 = time.time()
|
||||
return response.choices[0].message.content, t2 - t1
|
||||
|
||||
|
||||
def prompt(state):
|
||||
"""Prepare the messages for the LLM."""
|
||||
store = get_store()
|
||||
memories = store.search(
|
||||
("memories",),
|
||||
query=state["messages"][-1].content,
|
||||
)
|
||||
system_msg = f"""You are a helpful assistant.
|
||||
|
||||
## Memories
|
||||
<memories>
|
||||
{memories}
|
||||
</memories>
|
||||
"""
|
||||
return [{"role": "system", "content": system_msg}, *state["messages"]]
|
||||
|
||||
|
||||
class LangMem:
|
||||
def __init__(
|
||||
self,
|
||||
):
|
||||
self.store = InMemoryStore(
|
||||
index={
|
||||
"dims": 1536,
|
||||
"embed": f"openai:{os.getenv('EMBEDDING_MODEL')}",
|
||||
}
|
||||
)
|
||||
self.checkpointer = MemorySaver() # Checkpoint graph state
|
||||
|
||||
self.agent = create_react_agent(
|
||||
f"openai:{os.getenv('MODEL')}",
|
||||
prompt=prompt,
|
||||
tools=[
|
||||
create_manage_memory_tool(namespace=("memories",)),
|
||||
create_search_memory_tool(namespace=("memories",)),
|
||||
],
|
||||
store=self.store,
|
||||
checkpointer=self.checkpointer,
|
||||
)
|
||||
|
||||
def add_memory(self, message, config):
|
||||
return self.agent.invoke({"messages": [{"role": "user", "content": message}]}, config=config)
|
||||
|
||||
def search_memory(self, query, config):
|
||||
try:
|
||||
t1 = time.time()
|
||||
response = self.agent.invoke({"messages": [{"role": "user", "content": query}]}, config=config)
|
||||
t2 = time.time()
|
||||
return response["messages"][-1].content, t2 - t1
|
||||
except Exception as e:
|
||||
print(f"Error in search_memory: {e}")
|
||||
return "", t2 - t1
|
||||
|
||||
|
||||
class LangMemManager:
|
||||
def __init__(self, dataset_path):
|
||||
self.dataset_path = dataset_path
|
||||
with open(self.dataset_path, "r") as f:
|
||||
self.data = json.load(f)
|
||||
|
||||
def process_all_conversations(self, output_file_path):
|
||||
OUTPUT = defaultdict(list)
|
||||
|
||||
# Process conversations in parallel with multiple workers
|
||||
def process_conversation(key_value_pair):
|
||||
key, value = key_value_pair
|
||||
result = defaultdict(list)
|
||||
|
||||
chat_history = value["conversation"]
|
||||
questions = value["question"]
|
||||
|
||||
agent1 = LangMem()
|
||||
agent2 = LangMem()
|
||||
config = {"configurable": {"thread_id": f"thread-{key}"}}
|
||||
speakers = set()
|
||||
|
||||
# Identify speakers
|
||||
for conv in chat_history:
|
||||
speakers.add(conv["speaker"])
|
||||
|
||||
if len(speakers) != 2:
|
||||
raise ValueError(f"Expected 2 speakers, got {len(speakers)}")
|
||||
|
||||
speaker1 = list(speakers)[0]
|
||||
speaker2 = list(speakers)[1]
|
||||
|
||||
# Add memories for each message
|
||||
for conv in tqdm(chat_history, desc=f"Processing messages {key}", leave=False):
|
||||
message = f"{conv['timestamp']} | {conv['speaker']}: {conv['text']}"
|
||||
if conv["speaker"] == speaker1:
|
||||
agent1.add_memory(message, config)
|
||||
elif conv["speaker"] == speaker2:
|
||||
agent2.add_memory(message, config)
|
||||
else:
|
||||
raise ValueError(f"Expected speaker1 or speaker2, got {conv['speaker']}")
|
||||
|
||||
# Process questions
|
||||
for q in tqdm(questions, desc=f"Processing questions {key}", leave=False):
|
||||
category = q["category"]
|
||||
|
||||
if int(category) == 5:
|
||||
continue
|
||||
|
||||
answer = q["answer"]
|
||||
question = q["question"]
|
||||
response1, speaker1_memory_time = agent1.search_memory(question, config)
|
||||
response2, speaker2_memory_time = agent2.search_memory(question, config)
|
||||
|
||||
generated_answer, response_time = get_answer(question, speaker1, response1, speaker2, response2)
|
||||
|
||||
result[key].append(
|
||||
{
|
||||
"question": question,
|
||||
"answer": answer,
|
||||
"response1": response1,
|
||||
"response2": response2,
|
||||
"category": category,
|
||||
"speaker1_memory_time": speaker1_memory_time,
|
||||
"speaker2_memory_time": speaker2_memory_time,
|
||||
"response_time": response_time,
|
||||
"response": generated_answer,
|
||||
}
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
# Use multiprocessing to process conversations in parallel
|
||||
with mp.Pool(processes=10) as pool:
|
||||
results = list(
|
||||
tqdm(
|
||||
pool.imap(process_conversation, list(self.data.items())),
|
||||
total=len(self.data),
|
||||
desc="Processing conversations",
|
||||
)
|
||||
)
|
||||
|
||||
# Combine results from all workers
|
||||
for result in results:
|
||||
for key, items in result.items():
|
||||
OUTPUT[key].extend(items)
|
||||
|
||||
# Save final results
|
||||
with open(output_file_path, "w") as f:
|
||||
json.dump(OUTPUT, f, indent=4)
|
||||
@@ -1,141 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from tqdm import tqdm
|
||||
|
||||
from mem0 import MemoryClient
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
# Update custom instructions
|
||||
custom_instructions = """
|
||||
Generate personal memories that follow these guidelines:
|
||||
|
||||
1. Each memory should be self-contained with complete context, including:
|
||||
- The person's name, do not use "user" while creating memories
|
||||
- Personal details (career aspirations, hobbies, life circumstances)
|
||||
- Emotional states and reactions
|
||||
- Ongoing journeys or future plans
|
||||
- Specific dates when events occurred
|
||||
|
||||
2. Include meaningful personal narratives focusing on:
|
||||
- Identity and self-acceptance journeys
|
||||
- Family planning and parenting
|
||||
- Creative outlets and hobbies
|
||||
- Mental health and self-care activities
|
||||
- Career aspirations and education goals
|
||||
- Important life events and milestones
|
||||
|
||||
3. Make each memory rich with specific details rather than general statements
|
||||
- Include timeframes (exact dates when possible)
|
||||
- Name specific activities (e.g., "charity race for mental health" rather than just "exercise")
|
||||
- Include emotional context and personal growth elements
|
||||
|
||||
4. Extract memories only from user messages, not incorporating assistant responses
|
||||
|
||||
5. Format each memory as a paragraph with a clear narrative structure that captures the person's experience, challenges, and aspirations
|
||||
"""
|
||||
|
||||
|
||||
class MemoryADD:
|
||||
def __init__(self, data_path=None, batch_size=2, is_graph=False):
|
||||
self.mem0_client = MemoryClient(
|
||||
api_key=os.getenv("MEM0_API_KEY"),
|
||||
org_id=os.getenv("MEM0_ORGANIZATION_ID"),
|
||||
project_id=os.getenv("MEM0_PROJECT_ID"),
|
||||
)
|
||||
|
||||
self.mem0_client.update_project(custom_instructions=custom_instructions)
|
||||
self.batch_size = batch_size
|
||||
self.data_path = data_path
|
||||
self.data = None
|
||||
self.is_graph = is_graph
|
||||
if data_path:
|
||||
self.load_data()
|
||||
|
||||
def load_data(self):
|
||||
with open(self.data_path, "r") as f:
|
||||
self.data = json.load(f)
|
||||
return self.data
|
||||
|
||||
def add_memory(self, user_id, message, metadata, retries=3):
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
_ = self.mem0_client.add(
|
||||
message, user_id=user_id, version="v2", metadata=metadata, enable_graph=self.is_graph
|
||||
)
|
||||
return
|
||||
except Exception as e:
|
||||
if attempt < retries - 1:
|
||||
time.sleep(1) # Wait before retrying
|
||||
continue
|
||||
else:
|
||||
raise e
|
||||
|
||||
def add_memories_for_speaker(self, speaker, messages, timestamp, desc):
|
||||
for i in tqdm(range(0, len(messages), self.batch_size), desc=desc):
|
||||
batch_messages = messages[i : i + self.batch_size]
|
||||
self.add_memory(speaker, batch_messages, metadata={"timestamp": timestamp})
|
||||
|
||||
def process_conversation(self, item, idx):
|
||||
conversation = item["conversation"]
|
||||
speaker_a = conversation["speaker_a"]
|
||||
speaker_b = conversation["speaker_b"]
|
||||
|
||||
speaker_a_user_id = f"{speaker_a}_{idx}"
|
||||
speaker_b_user_id = f"{speaker_b}_{idx}"
|
||||
|
||||
# delete all memories for the two users
|
||||
self.mem0_client.delete_all(user_id=speaker_a_user_id)
|
||||
self.mem0_client.delete_all(user_id=speaker_b_user_id)
|
||||
|
||||
for key in conversation.keys():
|
||||
if key in ["speaker_a", "speaker_b"] or "date" in key or "timestamp" in key:
|
||||
continue
|
||||
|
||||
date_time_key = key + "_date_time"
|
||||
timestamp = conversation[date_time_key]
|
||||
chats = conversation[key]
|
||||
|
||||
messages = []
|
||||
messages_reverse = []
|
||||
for chat in chats:
|
||||
if chat["speaker"] == speaker_a:
|
||||
messages.append({"role": "user", "content": f"{speaker_a}: {chat['text']}"})
|
||||
messages_reverse.append({"role": "assistant", "content": f"{speaker_a}: {chat['text']}"})
|
||||
elif chat["speaker"] == speaker_b:
|
||||
messages.append({"role": "assistant", "content": f"{speaker_b}: {chat['text']}"})
|
||||
messages_reverse.append({"role": "user", "content": f"{speaker_b}: {chat['text']}"})
|
||||
else:
|
||||
raise ValueError(f"Unknown speaker: {chat['speaker']}")
|
||||
|
||||
# add memories for the two users on different threads
|
||||
thread_a = threading.Thread(
|
||||
target=self.add_memories_for_speaker,
|
||||
args=(speaker_a_user_id, messages, timestamp, "Adding Memories for Speaker A"),
|
||||
)
|
||||
thread_b = threading.Thread(
|
||||
target=self.add_memories_for_speaker,
|
||||
args=(speaker_b_user_id, messages_reverse, timestamp, "Adding Memories for Speaker B"),
|
||||
)
|
||||
|
||||
thread_a.start()
|
||||
thread_b.start()
|
||||
thread_a.join()
|
||||
thread_b.join()
|
||||
|
||||
print("Messages added successfully")
|
||||
|
||||
def process_all_conversations(self, max_workers=10):
|
||||
if not self.data:
|
||||
raise ValueError("No data loaded. Please set data_path and call load_data() first.")
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||
futures = [executor.submit(self.process_conversation, item, idx) for idx, item in enumerate(self.data)]
|
||||
|
||||
for future in futures:
|
||||
future.result()
|
||||
@@ -1,215 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from collections import defaultdict
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from jinja2 import Template
|
||||
from openai import OpenAI
|
||||
from prompts import ANSWER_PROMPT, ANSWER_PROMPT_GRAPH
|
||||
from tqdm import tqdm
|
||||
|
||||
from mem0 import MemoryClient
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
class MemorySearch:
|
||||
def __init__(self, output_path="results.json", top_k=10, filter_memories=False, is_graph=False):
|
||||
self.mem0_client = MemoryClient(
|
||||
api_key=os.getenv("MEM0_API_KEY"),
|
||||
org_id=os.getenv("MEM0_ORGANIZATION_ID"),
|
||||
project_id=os.getenv("MEM0_PROJECT_ID"),
|
||||
)
|
||||
self.top_k = top_k
|
||||
self.openai_client = OpenAI()
|
||||
self.results = defaultdict(list)
|
||||
self.output_path = output_path
|
||||
self.filter_memories = filter_memories
|
||||
self.is_graph = is_graph
|
||||
|
||||
if self.is_graph:
|
||||
self.ANSWER_PROMPT = ANSWER_PROMPT_GRAPH
|
||||
else:
|
||||
self.ANSWER_PROMPT = ANSWER_PROMPT
|
||||
|
||||
def search_memory(self, user_id, query, max_retries=3, retry_delay=1):
|
||||
start_time = time.time()
|
||||
retries = 0
|
||||
while retries < max_retries:
|
||||
try:
|
||||
if self.is_graph:
|
||||
print("Searching with graph")
|
||||
memories = self.mem0_client.search(
|
||||
query,
|
||||
user_id=user_id,
|
||||
top_k=self.top_k,
|
||||
filter_memories=self.filter_memories,
|
||||
enable_graph=True,
|
||||
output_format="v1.1",
|
||||
)
|
||||
else:
|
||||
memories = self.mem0_client.search(
|
||||
query, user_id=user_id, top_k=self.top_k, filter_memories=self.filter_memories
|
||||
)
|
||||
break
|
||||
except Exception as e:
|
||||
print("Retrying...")
|
||||
retries += 1
|
||||
if retries >= max_retries:
|
||||
raise e
|
||||
time.sleep(retry_delay)
|
||||
|
||||
end_time = time.time()
|
||||
if not self.is_graph:
|
||||
semantic_memories = [
|
||||
{
|
||||
"memory": memory["memory"],
|
||||
"timestamp": memory["metadata"]["timestamp"],
|
||||
"score": round(memory["score"], 2),
|
||||
}
|
||||
for memory in memories
|
||||
]
|
||||
graph_memories = None
|
||||
else:
|
||||
semantic_memories = [
|
||||
{
|
||||
"memory": memory["memory"],
|
||||
"timestamp": memory["metadata"]["timestamp"],
|
||||
"score": round(memory["score"], 2),
|
||||
}
|
||||
for memory in memories["results"]
|
||||
]
|
||||
graph_memories = [
|
||||
{"source": relation["source"], "relationship": relation["relationship"], "target": relation["target"]}
|
||||
for relation in memories["relations"]
|
||||
]
|
||||
return semantic_memories, graph_memories, end_time - start_time
|
||||
|
||||
def answer_question(self, speaker_1_user_id, speaker_2_user_id, question, answer, category):
|
||||
speaker_1_memories, speaker_1_graph_memories, speaker_1_memory_time = self.search_memory(
|
||||
speaker_1_user_id, question
|
||||
)
|
||||
speaker_2_memories, speaker_2_graph_memories, speaker_2_memory_time = self.search_memory(
|
||||
speaker_2_user_id, question
|
||||
)
|
||||
|
||||
search_1_memory = [f"{item['timestamp']}: {item['memory']}" for item in speaker_1_memories]
|
||||
search_2_memory = [f"{item['timestamp']}: {item['memory']}" for item in speaker_2_memories]
|
||||
|
||||
template = Template(self.ANSWER_PROMPT)
|
||||
answer_prompt = template.render(
|
||||
speaker_1_user_id=speaker_1_user_id.split("_")[0],
|
||||
speaker_2_user_id=speaker_2_user_id.split("_")[0],
|
||||
speaker_1_memories=json.dumps(search_1_memory, indent=4),
|
||||
speaker_2_memories=json.dumps(search_2_memory, indent=4),
|
||||
speaker_1_graph_memories=json.dumps(speaker_1_graph_memories, indent=4),
|
||||
speaker_2_graph_memories=json.dumps(speaker_2_graph_memories, indent=4),
|
||||
question=question,
|
||||
)
|
||||
|
||||
t1 = time.time()
|
||||
response = self.openai_client.chat.completions.create(
|
||||
model=os.getenv("MODEL"), messages=[{"role": "system", "content": answer_prompt}], temperature=0.0
|
||||
)
|
||||
t2 = time.time()
|
||||
response_time = t2 - t1
|
||||
return (
|
||||
response.choices[0].message.content,
|
||||
speaker_1_memories,
|
||||
speaker_2_memories,
|
||||
speaker_1_memory_time,
|
||||
speaker_2_memory_time,
|
||||
speaker_1_graph_memories,
|
||||
speaker_2_graph_memories,
|
||||
response_time,
|
||||
)
|
||||
|
||||
def process_question(self, val, speaker_a_user_id, speaker_b_user_id):
|
||||
question = val.get("question", "")
|
||||
answer = val.get("answer", "")
|
||||
category = val.get("category", -1)
|
||||
evidence = val.get("evidence", [])
|
||||
adversarial_answer = val.get("adversarial_answer", "")
|
||||
|
||||
(
|
||||
response,
|
||||
speaker_1_memories,
|
||||
speaker_2_memories,
|
||||
speaker_1_memory_time,
|
||||
speaker_2_memory_time,
|
||||
speaker_1_graph_memories,
|
||||
speaker_2_graph_memories,
|
||||
response_time,
|
||||
) = self.answer_question(speaker_a_user_id, speaker_b_user_id, question, answer, category)
|
||||
|
||||
result = {
|
||||
"question": question,
|
||||
"answer": answer,
|
||||
"category": category,
|
||||
"evidence": evidence,
|
||||
"response": response,
|
||||
"adversarial_answer": adversarial_answer,
|
||||
"speaker_1_memories": speaker_1_memories,
|
||||
"speaker_2_memories": speaker_2_memories,
|
||||
"num_speaker_1_memories": len(speaker_1_memories),
|
||||
"num_speaker_2_memories": len(speaker_2_memories),
|
||||
"speaker_1_memory_time": speaker_1_memory_time,
|
||||
"speaker_2_memory_time": speaker_2_memory_time,
|
||||
"speaker_1_graph_memories": speaker_1_graph_memories,
|
||||
"speaker_2_graph_memories": speaker_2_graph_memories,
|
||||
"response_time": response_time,
|
||||
}
|
||||
|
||||
# Save results after each question is processed
|
||||
with open(self.output_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
return result
|
||||
|
||||
def process_data_file(self, file_path):
|
||||
with open(file_path, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
for idx, item in tqdm(enumerate(data), total=len(data), desc="Processing conversations"):
|
||||
qa = item["qa"]
|
||||
conversation = item["conversation"]
|
||||
speaker_a = conversation["speaker_a"]
|
||||
speaker_b = conversation["speaker_b"]
|
||||
|
||||
speaker_a_user_id = f"{speaker_a}_{idx}"
|
||||
speaker_b_user_id = f"{speaker_b}_{idx}"
|
||||
|
||||
for question_item in tqdm(
|
||||
qa, total=len(qa), desc=f"Processing questions for conversation {idx}", leave=False
|
||||
):
|
||||
result = self.process_question(question_item, speaker_a_user_id, speaker_b_user_id)
|
||||
self.results[idx].append(result)
|
||||
|
||||
# Save results after each question is processed
|
||||
with open(self.output_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
# Final save at the end
|
||||
with open(self.output_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
def process_questions_parallel(self, qa_list, speaker_a_user_id, speaker_b_user_id, max_workers=1):
|
||||
def process_single_question(val):
|
||||
result = self.process_question(val, speaker_a_user_id, speaker_b_user_id)
|
||||
# Save results after each question is processed
|
||||
with open(self.output_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
return result
|
||||
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||
results = list(
|
||||
tqdm(executor.map(process_single_question, qa_list), total=len(qa_list), desc="Answering Questions")
|
||||
)
|
||||
|
||||
# Final save at the end
|
||||
with open(self.output_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
return results
|
||||
@@ -1,131 +0,0 @@
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from collections import defaultdict
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from jinja2 import Template
|
||||
from openai import OpenAI
|
||||
from tqdm import tqdm
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
ANSWER_PROMPT = """
|
||||
You are an intelligent memory assistant tasked with retrieving accurate information from conversation memories.
|
||||
|
||||
# CONTEXT:
|
||||
You have access to memories from a conversation. These memories contain
|
||||
timestamped information that may be relevant to answering the question.
|
||||
|
||||
# INSTRUCTIONS:
|
||||
1. Carefully analyze all provided memories
|
||||
2. Pay special attention to the timestamps to determine the answer
|
||||
3. If the question asks about a specific event or fact, look for direct evidence in the memories
|
||||
4. If the memories contain contradictory information, prioritize the most recent memory
|
||||
5. If there is a question about time references (like "last year", "two months ago", etc.),
|
||||
calculate the actual date based on the memory timestamp. For example, if a memory from
|
||||
4 May 2022 mentions "went to India last year," then the trip occurred in 2021.
|
||||
6. Always convert relative time references to specific dates, months, or years. For example,
|
||||
convert "last year" to "2022" or "two months ago" to "March 2023" based on the memory
|
||||
timestamp. Ignore the reference while answering the question.
|
||||
7. Focus only on the content of the memories. Do not confuse character
|
||||
names mentioned in memories with the actual users who created those memories.
|
||||
8. The answer should be less than 5-6 words.
|
||||
|
||||
# APPROACH (Think step by step):
|
||||
1. First, examine all memories that contain information related to the question
|
||||
2. Examine the timestamps and content of these memories carefully
|
||||
3. Look for explicit mentions of dates, times, locations, or events that answer the question
|
||||
4. If the answer requires calculation (e.g., converting relative time references), show your work
|
||||
5. Formulate a precise, concise answer based solely on the evidence in the memories
|
||||
6. Double-check that your answer directly addresses the question asked
|
||||
7. Ensure your final answer is specific and avoids vague time references
|
||||
|
||||
Memories:
|
||||
|
||||
{{memories}}
|
||||
|
||||
Question: {{question}}
|
||||
Answer:
|
||||
"""
|
||||
|
||||
|
||||
class OpenAIPredict:
|
||||
def __init__(self, model="gpt-4o-mini"):
|
||||
self.model = model
|
||||
self.openai_client = OpenAI()
|
||||
self.results = defaultdict(list)
|
||||
|
||||
def search_memory(self, idx):
|
||||
with open(f"memories/{idx}.txt", "r") as file:
|
||||
memories = file.read()
|
||||
|
||||
return memories, 0
|
||||
|
||||
def process_question(self, val, idx):
|
||||
question = val.get("question", "")
|
||||
answer = val.get("answer", "")
|
||||
category = val.get("category", -1)
|
||||
evidence = val.get("evidence", [])
|
||||
adversarial_answer = val.get("adversarial_answer", "")
|
||||
|
||||
response, search_memory_time, response_time, context = self.answer_question(idx, question)
|
||||
|
||||
result = {
|
||||
"question": question,
|
||||
"answer": answer,
|
||||
"category": category,
|
||||
"evidence": evidence,
|
||||
"response": response,
|
||||
"adversarial_answer": adversarial_answer,
|
||||
"search_memory_time": search_memory_time,
|
||||
"response_time": response_time,
|
||||
"context": context,
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
def answer_question(self, idx, question):
|
||||
memories, search_memory_time = self.search_memory(idx)
|
||||
|
||||
template = Template(ANSWER_PROMPT)
|
||||
answer_prompt = template.render(memories=memories, question=question)
|
||||
|
||||
t1 = time.time()
|
||||
response = self.openai_client.chat.completions.create(
|
||||
model=os.getenv("MODEL"), messages=[{"role": "system", "content": answer_prompt}], temperature=0.0
|
||||
)
|
||||
t2 = time.time()
|
||||
response_time = t2 - t1
|
||||
return response.choices[0].message.content, search_memory_time, response_time, memories
|
||||
|
||||
def process_data_file(self, file_path, output_file_path):
|
||||
with open(file_path, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
for idx, item in tqdm(enumerate(data), total=len(data), desc="Processing conversations"):
|
||||
qa = item["qa"]
|
||||
|
||||
for question_item in tqdm(
|
||||
qa, total=len(qa), desc=f"Processing questions for conversation {idx}", leave=False
|
||||
):
|
||||
result = self.process_question(question_item, idx)
|
||||
self.results[idx].append(result)
|
||||
|
||||
# Save results after each question is processed
|
||||
with open(output_file_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
# Final save at the end
|
||||
with open(output_file_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--output_file_path", type=str, required=True)
|
||||
args = parser.parse_args()
|
||||
openai_predict = OpenAIPredict()
|
||||
openai_predict.process_data_file("../../dataset/locomo10.json", args.output_file_path)
|
||||
@@ -1,183 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from collections import defaultdict
|
||||
|
||||
import numpy as np
|
||||
import tiktoken
|
||||
from dotenv import load_dotenv
|
||||
from jinja2 import Template
|
||||
from openai import OpenAI
|
||||
from tqdm import tqdm
|
||||
|
||||
load_dotenv()
|
||||
|
||||
PROMPT = """
|
||||
# Question:
|
||||
{{QUESTION}}
|
||||
|
||||
# Context:
|
||||
{{CONTEXT}}
|
||||
|
||||
# Short answer:
|
||||
"""
|
||||
|
||||
|
||||
class RAGManager:
|
||||
def __init__(self, data_path="dataset/locomo10_rag.json", chunk_size=500, k=1):
|
||||
self.model = os.getenv("MODEL")
|
||||
self.client = OpenAI()
|
||||
self.data_path = data_path
|
||||
self.chunk_size = chunk_size
|
||||
self.k = k
|
||||
|
||||
def generate_response(self, question, context):
|
||||
template = Template(PROMPT)
|
||||
prompt = template.render(CONTEXT=context, QUESTION=question)
|
||||
|
||||
max_retries = 3
|
||||
retries = 0
|
||||
|
||||
while retries <= max_retries:
|
||||
try:
|
||||
t1 = time.time()
|
||||
response = self.client.chat.completions.create(
|
||||
model=self.model,
|
||||
messages=[
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You are a helpful assistant that can answer "
|
||||
"questions based on the provided context."
|
||||
"If the question involves timing, use the conversation date for reference."
|
||||
"Provide the shortest possible answer."
|
||||
"Use words directly from the conversation when possible."
|
||||
"Avoid using subjects in your answer.",
|
||||
},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
temperature=0,
|
||||
)
|
||||
t2 = time.time()
|
||||
return response.choices[0].message.content.strip(), t2 - t1
|
||||
except Exception as e:
|
||||
retries += 1
|
||||
if retries > max_retries:
|
||||
raise e
|
||||
time.sleep(1) # Wait before retrying
|
||||
|
||||
def clean_chat_history(self, chat_history):
|
||||
cleaned_chat_history = ""
|
||||
for c in chat_history:
|
||||
cleaned_chat_history += f"{c['timestamp']} | {c['speaker']}: {c['text']}\n"
|
||||
|
||||
return cleaned_chat_history
|
||||
|
||||
def calculate_embedding(self, document):
|
||||
response = self.client.embeddings.create(model=os.getenv("EMBEDDING_MODEL"), input=document)
|
||||
return response.data[0].embedding
|
||||
|
||||
def calculate_similarity(self, embedding1, embedding2):
|
||||
return np.dot(embedding1, embedding2) / (np.linalg.norm(embedding1) * np.linalg.norm(embedding2))
|
||||
|
||||
def search(self, query, chunks, embeddings, k=1):
|
||||
"""
|
||||
Search for the top-k most similar chunks to the query.
|
||||
|
||||
Args:
|
||||
query: The query string
|
||||
chunks: List of text chunks
|
||||
embeddings: List of embeddings for each chunk
|
||||
k: Number of top chunks to return (default: 1)
|
||||
|
||||
Returns:
|
||||
combined_chunks: The combined text of the top-k chunks
|
||||
search_time: Time taken for the search
|
||||
"""
|
||||
t1 = time.time()
|
||||
query_embedding = self.calculate_embedding(query)
|
||||
similarities = [self.calculate_similarity(query_embedding, embedding) for embedding in embeddings]
|
||||
|
||||
# Get indices of top-k most similar chunks
|
||||
if k == 1:
|
||||
# Original behavior - just get the most similar chunk
|
||||
top_indices = [np.argmax(similarities)]
|
||||
else:
|
||||
# Get indices of top-k chunks
|
||||
top_indices = np.argsort(similarities)[-k:][::-1]
|
||||
|
||||
# Combine the top-k chunks
|
||||
combined_chunks = "\n<->\n".join([chunks[i] for i in top_indices])
|
||||
|
||||
t2 = time.time()
|
||||
return combined_chunks, t2 - t1
|
||||
|
||||
def create_chunks(self, chat_history, chunk_size=500):
|
||||
"""
|
||||
Create chunks using tiktoken for more accurate token counting
|
||||
"""
|
||||
# Get the encoding for the model
|
||||
encoding = tiktoken.encoding_for_model(os.getenv("EMBEDDING_MODEL"))
|
||||
|
||||
documents = self.clean_chat_history(chat_history)
|
||||
|
||||
if chunk_size == -1:
|
||||
return [documents], []
|
||||
|
||||
chunks = []
|
||||
|
||||
# Encode the document
|
||||
tokens = encoding.encode(documents)
|
||||
|
||||
# Split into chunks based on token count
|
||||
for i in range(0, len(tokens), chunk_size):
|
||||
chunk_tokens = tokens[i : i + chunk_size]
|
||||
chunk = encoding.decode(chunk_tokens)
|
||||
chunks.append(chunk)
|
||||
|
||||
embeddings = []
|
||||
for chunk in chunks:
|
||||
embedding = self.calculate_embedding(chunk)
|
||||
embeddings.append(embedding)
|
||||
|
||||
return chunks, embeddings
|
||||
|
||||
def process_all_conversations(self, output_file_path):
|
||||
with open(self.data_path, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
FINAL_RESULTS = defaultdict(list)
|
||||
for key, value in tqdm(data.items(), desc="Processing conversations"):
|
||||
chat_history = value["conversation"]
|
||||
questions = value["question"]
|
||||
|
||||
chunks, embeddings = self.create_chunks(chat_history, self.chunk_size)
|
||||
|
||||
for item in tqdm(questions, desc="Answering questions", leave=False):
|
||||
question = item["question"]
|
||||
answer = item.get("answer", "")
|
||||
category = item["category"]
|
||||
|
||||
if self.chunk_size == -1:
|
||||
context = chunks[0]
|
||||
search_time = 0
|
||||
else:
|
||||
context, search_time = self.search(question, chunks, embeddings, k=self.k)
|
||||
response, response_time = self.generate_response(question, context)
|
||||
|
||||
FINAL_RESULTS[key].append(
|
||||
{
|
||||
"question": question,
|
||||
"answer": answer,
|
||||
"category": category,
|
||||
"context": context,
|
||||
"response": response,
|
||||
"search_time": search_time,
|
||||
"response_time": response_time,
|
||||
}
|
||||
)
|
||||
with open(output_file_path, "w+") as f:
|
||||
json.dump(FINAL_RESULTS, f, indent=4)
|
||||
|
||||
# Save results
|
||||
with open(output_file_path, "w+") as f:
|
||||
json.dump(FINAL_RESULTS, f, indent=4)
|
||||
@@ -1,3 +0,0 @@
|
||||
TECHNIQUES = ["mem0", "rag", "langmem", "zep", "openai"]
|
||||
|
||||
METHODS = ["add", "search"]
|
||||
@@ -1,76 +0,0 @@
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from tqdm import tqdm
|
||||
from zep_cloud import Message
|
||||
from zep_cloud.client import Zep
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
class ZepAdd:
|
||||
def __init__(self, data_path=None):
|
||||
self.zep_client = Zep(api_key=os.getenv("ZEP_API_KEY"))
|
||||
self.data_path = data_path
|
||||
self.data = None
|
||||
if data_path:
|
||||
self.load_data()
|
||||
|
||||
def load_data(self):
|
||||
with open(self.data_path, "r") as f:
|
||||
self.data = json.load(f)
|
||||
return self.data
|
||||
|
||||
def process_conversation(self, run_id, item, idx):
|
||||
conversation = item["conversation"]
|
||||
|
||||
user_id = f"run_id_{run_id}_experiment_user_{idx}"
|
||||
session_id = f"run_id_{run_id}_experiment_session_{idx}"
|
||||
|
||||
# # delete all memories for the two users
|
||||
# self.zep_client.user.delete(user_id=user_id)
|
||||
# self.zep_client.memory.delete(session_id=session_id)
|
||||
|
||||
self.zep_client.user.add(user_id=user_id)
|
||||
self.zep_client.memory.add_session(
|
||||
user_id=user_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
|
||||
print("Starting to add memories... for user", user_id)
|
||||
for key in tqdm(conversation.keys(), desc=f"Processing user {user_id}"):
|
||||
if key in ["speaker_a", "speaker_b"] or "date" in key:
|
||||
continue
|
||||
|
||||
date_time_key = key + "_date_time"
|
||||
timestamp = conversation[date_time_key]
|
||||
chats = conversation[key]
|
||||
|
||||
for chat in tqdm(chats, desc=f"Adding chats for {key}", leave=False):
|
||||
self.zep_client.memory.add(
|
||||
session_id=session_id,
|
||||
messages=[
|
||||
Message(
|
||||
role=chat["speaker"],
|
||||
role_type="user",
|
||||
content=f"{timestamp}: {chat['text']}",
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
def process_all_conversations(self, run_id):
|
||||
if not self.data:
|
||||
raise ValueError("No data loaded. Please set data_path and call load_data() first.")
|
||||
for idx, item in tqdm(enumerate(self.data)):
|
||||
if idx == 0:
|
||||
self.process_conversation(run_id, item, idx)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--run_id", type=str, required=True)
|
||||
args = parser.parse_args()
|
||||
zep_add = ZepAdd(data_path="../../dataset/locomo10.json")
|
||||
zep_add.process_all_conversations(args.run_id)
|
||||
@@ -1,140 +0,0 @@
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from collections import defaultdict
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from jinja2 import Template
|
||||
from openai import OpenAI
|
||||
from prompts import ANSWER_PROMPT_ZEP
|
||||
from tqdm import tqdm
|
||||
from zep_cloud import EntityEdge, EntityNode
|
||||
from zep_cloud.client import Zep
|
||||
|
||||
load_dotenv()
|
||||
|
||||
TEMPLATE = """
|
||||
FACTS and ENTITIES represent relevant context to the current conversation.
|
||||
|
||||
# These are the most relevant facts and their valid date ranges
|
||||
# format: FACT (Date range: from - to)
|
||||
|
||||
{facts}
|
||||
|
||||
|
||||
# These are the most relevant entities
|
||||
# ENTITY_NAME: entity summary
|
||||
|
||||
{entities}
|
||||
|
||||
"""
|
||||
|
||||
|
||||
class ZepSearch:
|
||||
def __init__(self):
|
||||
self.zep_client = Zep(api_key=os.getenv("ZEP_API_KEY"))
|
||||
self.results = defaultdict(list)
|
||||
self.openai_client = OpenAI()
|
||||
|
||||
def format_edge_date_range(self, edge: EntityEdge) -> str:
|
||||
# return f"{datetime(edge.valid_at).strftime('%Y-%m-%d %H:%M:%S') if edge.valid_at else 'date unknown'} - {(edge.invalid_at.strftime('%Y-%m-%d %H:%M:%S') if edge.invalid_at else 'present')}"
|
||||
return f"{edge.valid_at if edge.valid_at else 'date unknown'} - {(edge.invalid_at if edge.invalid_at else 'present')}"
|
||||
|
||||
def compose_search_context(self, edges: list[EntityEdge], nodes: list[EntityNode]) -> str:
|
||||
facts = [f" - {edge.fact} ({self.format_edge_date_range(edge)})" for edge in edges]
|
||||
entities = [f" - {node.name}: {node.summary}" for node in nodes]
|
||||
return TEMPLATE.format(facts="\n".join(facts), entities="\n".join(entities))
|
||||
|
||||
def search_memory(self, run_id, idx, query, max_retries=3, retry_delay=1):
|
||||
start_time = time.time()
|
||||
retries = 0
|
||||
while retries < max_retries:
|
||||
try:
|
||||
user_id = f"run_id_{run_id}_experiment_user_{idx}"
|
||||
edges_results = (
|
||||
self.zep_client.graph.search(
|
||||
user_id=user_id, reranker="cross_encoder", query=query, scope="edges", limit=20
|
||||
)
|
||||
).edges
|
||||
node_results = (
|
||||
self.zep_client.graph.search(user_id=user_id, reranker="rrf", query=query, scope="nodes", limit=20)
|
||||
).nodes
|
||||
context = self.compose_search_context(edges_results, node_results)
|
||||
break
|
||||
except Exception as e:
|
||||
print("Retrying...")
|
||||
retries += 1
|
||||
if retries >= max_retries:
|
||||
raise e
|
||||
time.sleep(retry_delay)
|
||||
|
||||
end_time = time.time()
|
||||
|
||||
return context, end_time - start_time
|
||||
|
||||
def process_question(self, run_id, val, idx):
|
||||
question = val.get("question", "")
|
||||
answer = val.get("answer", "")
|
||||
category = val.get("category", -1)
|
||||
evidence = val.get("evidence", [])
|
||||
adversarial_answer = val.get("adversarial_answer", "")
|
||||
|
||||
response, search_memory_time, response_time, context = self.answer_question(run_id, idx, question)
|
||||
|
||||
result = {
|
||||
"question": question,
|
||||
"answer": answer,
|
||||
"category": category,
|
||||
"evidence": evidence,
|
||||
"response": response,
|
||||
"adversarial_answer": adversarial_answer,
|
||||
"search_memory_time": search_memory_time,
|
||||
"response_time": response_time,
|
||||
"context": context,
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
def answer_question(self, run_id, idx, question):
|
||||
context, search_memory_time = self.search_memory(run_id, idx, question)
|
||||
|
||||
template = Template(ANSWER_PROMPT_ZEP)
|
||||
answer_prompt = template.render(memories=context, question=question)
|
||||
|
||||
t1 = time.time()
|
||||
response = self.openai_client.chat.completions.create(
|
||||
model=os.getenv("MODEL"), messages=[{"role": "system", "content": answer_prompt}], temperature=0.0
|
||||
)
|
||||
t2 = time.time()
|
||||
response_time = t2 - t1
|
||||
return response.choices[0].message.content, search_memory_time, response_time, context
|
||||
|
||||
def process_data_file(self, file_path, run_id, output_file_path):
|
||||
with open(file_path, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
for idx, item in tqdm(enumerate(data), total=len(data), desc="Processing conversations"):
|
||||
qa = item["qa"]
|
||||
|
||||
for question_item in tqdm(
|
||||
qa, total=len(qa), desc=f"Processing questions for conversation {idx}", leave=False
|
||||
):
|
||||
result = self.process_question(run_id, question_item, idx)
|
||||
self.results[idx].append(result)
|
||||
|
||||
# Save results after each question is processed
|
||||
with open(output_file_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
# Final save at the end
|
||||
with open(output_file_path, "w") as f:
|
||||
json.dump(self.results, f, indent=4)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--run_id", type=str, required=True)
|
||||
args = parser.parse_args()
|
||||
zep_search = ZepSearch()
|
||||
zep_search.process_data_file("../../dataset/locomo10.json", args.run_id, "results/zep_search_results.json")
|
||||
@@ -555,7 +555,7 @@
|
||||
"# - Enables creation of AI agents with long-term memory and learning abilities.\n",
|
||||
"# - Improves consistency and reduces repetition in user-agent interactions.\n",
|
||||
"\n",
|
||||
"from cookbooks.helper.mem0_teachability import Mem0Teachability\n",
|
||||
"from helper.mem0_teachability import Mem0Teachability\n",
|
||||
"\n",
|
||||
"teachability = Mem0Teachability(\n",
|
||||
" verbosity=2, # for visibility of what's happening\n",
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0",
|
||||
"version": "0.2.9",
|
||||
"version": "0.2.10",
|
||||
"description": "Persistent memory for Claude Code. Remembers decisions, patterns, and preferences across sessions.",
|
||||
"author": {
|
||||
"name": "Mem0",
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0",
|
||||
"version": "0.2.9",
|
||||
"version": "0.2.10",
|
||||
"description": "Persistent memory for Codex. Remembers decisions, patterns, and preferences across sessions.",
|
||||
"author": {
|
||||
"name": "Mem0",
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0",
|
||||
"version": "0.2.9",
|
||||
"version": "0.2.10",
|
||||
"description": "Mem0 memory layer for AI applications. Add persistent memory, personalization, and semantic search using the Mem0 Platform MCP server.",
|
||||
"author": {
|
||||
"name": "Mem0",
|
||||
+2
-1
@@ -2,13 +2,14 @@
|
||||
|
||||
All notable changes to the `@mem0/opencode-plugin` will be documented in this file.
|
||||
|
||||
## 0.1.3 — File-context injection, session summaries & activity timeline
|
||||
## 0.1.3 — File-context injection, session summaries & activity timeline, anonymous telemetry
|
||||
|
||||
### Added
|
||||
|
||||
- **File-context injection (`tool.execute.before` / Read):** Before the agent reads a file, the plugin searches mem0 for memories referencing that file path and injects prior work as system context. Gates on file size (>= 1,500 bytes). Gives the agent "I've worked on this file before" awareness automatically.
|
||||
- **Stop hook session summary (`experimental.session.compacting`):** Enhanced session compaction to store a structured `session_summary` memory with `infer=True`, letting the mem0 backend AI extract key facts (request, decisions, learnings, next steps). Previously only stored a raw stats string.
|
||||
- **SessionStart activity timeline:** The initial memory loading now formats recent memories with type icons (⚖️ decision, 🔴 bug_fix, 🔵 task_learning, etc.) and relative age indicators (2h ago, 1d ago) instead of bare text. Provides a visual "Recent Activity" timeline on first message.
|
||||
- **PostHog telemetry (`telemetry.ts`):** Anonymous, fire-and-forget usage events. Opt out with `MEM0_TELEMETRY=false`. Only fires when an API key is present; never sends memory content, prompts, or the API key — only an anonymized `sha256(apiKey)[:32]` identity plus event type, platform, and plugin version. Emits the same schema as the Mem0 editor plugin (`plugin.*` events, `source: "plugin"`, `platform: "opencode"`) so OpenCode appears as a `platform` in the shared plugin dashboard. Events: `plugin.session_start` (with memory count) and `plugin.tool_use` (`add` / `search` / `update` / `delete`).
|
||||
|
||||
### Changed
|
||||
|
||||
+1
-1
@@ -17,7 +17,7 @@ opencode plugin @mem0/opencode-plugin
|
||||
**Or let your agent do it** — paste this into OpenCode:
|
||||
|
||||
```
|
||||
Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/mem0-plugin/.opencode-plugin/README.md
|
||||
Install @mem0/opencode-plugin by following https://raw.githubusercontent.com/mem0ai/mem0/main/integrations/mem0-plugin/.opencode-plugin/README.md
|
||||
```
|
||||
|
||||
All commands auto-add the plugin and MCP server to your `~/.config/opencode/opencode.json`. No manual config needed.
|
||||
+2
-2
@@ -6,7 +6,7 @@
|
||||
"name": "@mem0/opencode-plugin",
|
||||
"dependencies": {
|
||||
"@opencode-ai/plugin": "^1.0.162",
|
||||
"mem0ai": "^3.0.5",
|
||||
"mem0ai": "^3.0.7",
|
||||
},
|
||||
"devDependencies": {
|
||||
"bun-types": ">=1.3.14",
|
||||
@@ -462,7 +462,7 @@
|
||||
|
||||
"md5": ["md5@2.3.0", "", { "dependencies": { "charenc": "0.0.2", "crypt": "0.0.2", "is-buffer": "~1.1.6" } }, "sha512-T1GITYmFaKuO91vxyoQMFETst+O71VUPEU3ze5GNzDm0OWdP8v1ziTaAEPUr/3kLsY3Sftgz242A1SetQiDL7g=="],
|
||||
|
||||
"mem0ai": ["mem0ai@3.0.5", "", { "dependencies": { "axios": "^1.15.2", "openai": "^4.93.0", "uuid": "9.0.1", "zod": "^3.24.1" }, "peerDependencies": { "@anthropic-ai/sdk": "^0.40.1", "@azure/identity": "^4.0.0", "@azure/search-documents": "^12.0.0", "@cloudflare/workers-types": "^4.20250504.0", "@google/genai": "^1.2.0", "@langchain/core": "^1.1.47", "@mistralai/mistralai": "^1.5.2", "@qdrant/js-client-rest": "1.13.0", "@supabase/supabase-js": "^2.49.1", "@types/jest": "29.5.14", "@types/pg": "8.11.0", "better-sqlite3": "^12.6.2", "cloudflare": "^4.2.0", "compromise": "^14.0.0", "groq-sdk": "0.3.0", "natural": "^8.0.1", "ollama": "^0.5.14", "pg": "8.11.3", "redis": "^4.6.13" } }, "sha512-W/R59d5fMpUGHhPEnyoo36GSz5NFJbAs+vS4BxoIvE+t19mIJfoz/2FJSKIw80mT8AkeNYpDfzc/DYRu2IzYIw=="],
|
||||
"mem0ai": ["mem0ai@3.0.7", "", { "dependencies": { "axios": "^1.16.0", "openai": "^4.93.0", "uuid": "9.0.1", "zod": "^3.24.1" }, "peerDependencies": { "@anthropic-ai/sdk": "^0.40.1", "@azure/identity": "^4.0.0", "@azure/search-documents": "^12.0.0", "@cloudflare/workers-types": "^4.20250504.0", "@google/genai": "^1.40.0", "@langchain/core": "^1.1.47", "@mistralai/mistralai": "^1.5.2", "@qdrant/js-client-rest": "^1.18.0", "@supabase/supabase-js": "^2.49.1", "@types/jest": "29.5.14", "@types/pg": "8.11.0", "better-sqlite3": "^12.6.2", "cloudflare": "^4.2.0", "compromise": "^14.0.0", "groq-sdk": "0.3.0", "natural": "^8.0.1", "ollama": "^0.5.14", "pg": "8.11.3", "redis": "^4.6.13" } }, "sha512-CUHzX7DyeKTHcI3aDsSqY9LXTD7GcFxf988796TuOa4yJgGuF2Xd2NROcBhVFRo3r9y8fVmbo3c5TF9jv1KlKw=="],
|
||||
|
||||
"memjs": ["memjs@1.3.2", "", {}, "sha512-qUEg2g8vxPe+zPn09KidjIStHPtoBO8Cttm8bgJFWWabbsjQ9Av9Ky+6UcvKx6ue0LLb/LEhtcyQpRyKfzeXcg=="],
|
||||
|
||||
+14
@@ -9,6 +9,7 @@ import { existsSync, readdirSync, cpSync, mkdirSync, readFileSync, writeFileSync
|
||||
import { homedir } from "os";
|
||||
import { join } from "path";
|
||||
import { createHash } from "crypto";
|
||||
import { captureEvent } from "./telemetry";
|
||||
|
||||
async function getUserId(): Promise<string> {
|
||||
if (process.env.MEM0_USER_ID) return process.env.MEM0_USER_ID;
|
||||
@@ -368,6 +369,8 @@ const Mem0Plugin: Plugin = async (ctx) => {
|
||||
});
|
||||
} catch {}
|
||||
}
|
||||
|
||||
captureEvent("session_start", { memory_count: memoryCount }, apiKey);
|
||||
}
|
||||
|
||||
if (NUDGE_RE.test(safeText)) {
|
||||
@@ -602,6 +605,17 @@ const Mem0Plugin: Plugin = async (ctx) => {
|
||||
if (MEM0_MCP_RE.test(toolName)) {
|
||||
if (toolName.includes("add_memory")) stats.adds++;
|
||||
if (toolName.includes("search")) stats.searches++;
|
||||
|
||||
const tool = toolName.includes("add_memory")
|
||||
? "add_memory"
|
||||
: toolName.includes("search")
|
||||
? "search_memories"
|
||||
: toolName.includes("delete")
|
||||
? "delete_memory"
|
||||
: toolName.includes("update")
|
||||
? "update_memory"
|
||||
: "other";
|
||||
captureEvent("tool_use", { tool }, apiKey);
|
||||
}
|
||||
|
||||
if (toolName === "bash" && toolOutput.length >= 50) {
|
||||
+2
-2
@@ -30,7 +30,7 @@
|
||||
"repository": {
|
||||
"type": "git",
|
||||
"url": "https://github.com/mem0ai/mem0",
|
||||
"directory": "mem0-plugin/.opencode-plugin"
|
||||
"directory": "integrations/mem0-plugin/.opencode-plugin"
|
||||
},
|
||||
"files": [
|
||||
"dist",
|
||||
@@ -59,7 +59,7 @@
|
||||
},
|
||||
"dependencies": {
|
||||
"@opencode-ai/plugin": "^1.0.162",
|
||||
"mem0ai": "^3.0.5"
|
||||
"mem0ai": "^3.0.7"
|
||||
},
|
||||
"devDependencies": {
|
||||
"bun-types": ">=1.3.14",
|
||||
@@ -0,0 +1,51 @@
|
||||
import { afterEach, describe, expect, test } from "bun:test";
|
||||
import { buildEvent, captureEvent, isTelemetryEnabled } from "./telemetry";
|
||||
|
||||
const KEY = "m0-testkey123";
|
||||
|
||||
afterEach(() => {
|
||||
delete process.env.MEM0_TELEMETRY;
|
||||
});
|
||||
|
||||
describe("opencode telemetry", () => {
|
||||
test("buildEvent uses the shared plugin.* schema with platform=opencode", () => {
|
||||
const payload = buildEvent("session_start", { memory_count: 5 }, KEY);
|
||||
expect(payload).not.toBeNull();
|
||||
const props = payload!.properties as Record<string, unknown>;
|
||||
expect(payload!.event).toBe("plugin.session_start");
|
||||
expect(props.source).toBe("plugin");
|
||||
expect(props.platform).toBe("opencode");
|
||||
expect(props.memory_count).toBe(5);
|
||||
expect(props.$process_person_profile).toBe(false);
|
||||
expect(typeof props.plugin_version).toBe("string");
|
||||
});
|
||||
|
||||
test("distinct_id is sha256(apiKey)[:32] — matches the editor plugin", async () => {
|
||||
const { createHash } = await import("node:crypto");
|
||||
const expected = createHash("sha256").update(KEY).digest("hex").slice(0, 32);
|
||||
expect(buildEvent("session_start", {}, KEY)!.distinct_id).toBe(expected);
|
||||
});
|
||||
|
||||
test("system properties win over caller-supplied ones", () => {
|
||||
const props = buildEvent("x", { platform: "HACK", source: "HACK" }, KEY)!
|
||||
.properties as Record<string, unknown>;
|
||||
expect(props.platform).toBe("opencode");
|
||||
expect(props.source).toBe("plugin");
|
||||
});
|
||||
|
||||
test("returns null without an API key (no anonymous events)", () => {
|
||||
expect(buildEvent("session_start", {}, undefined)).toBeNull();
|
||||
});
|
||||
|
||||
test("opt-out via MEM0_TELEMETRY disables events", () => {
|
||||
process.env.MEM0_TELEMETRY = "false";
|
||||
expect(isTelemetryEnabled()).toBe(false);
|
||||
expect(buildEvent("session_start", {}, KEY)).toBeNull();
|
||||
});
|
||||
|
||||
test("captureEvent never throws (and sends nothing when opted out)", () => {
|
||||
process.env.MEM0_TELEMETRY = "false";
|
||||
expect(() => captureEvent("session_start", {}, KEY)).not.toThrow();
|
||||
expect(() => captureEvent("session_start", {}, undefined)).not.toThrow();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,104 @@
|
||||
/**
|
||||
* Plugin telemetry for the Mem0 OpenCode plugin — anonymous usage tracking
|
||||
* via PostHog.
|
||||
*
|
||||
* Emits the SAME event schema as the Mem0 editor plugin's telemetry.py
|
||||
* (event names prefixed `plugin.`, `source: "plugin"`, `platform: "opencode"`,
|
||||
* `distinct_id = sha256(apiKey)[:32]`) so OpenCode shows up as just another
|
||||
* `platform` value in the shared plugin dashboard instead of a separate
|
||||
* event namespace.
|
||||
*
|
||||
* Fire-and-forget: never throws, never blocks, failures are swallowed. Only
|
||||
* fires when an API key is present (same as the editor plugin — anonymous
|
||||
* installs without a key emit nothing). Disable with MEM0_TELEMETRY=false.
|
||||
*
|
||||
* Never sends: memory content, API keys, raw user/project IDs. Only sends:
|
||||
* event type, platform, plugin version, anonymized hash of the API key.
|
||||
*/
|
||||
|
||||
import { createHash } from "node:crypto";
|
||||
import { readFileSync } from "node:fs";
|
||||
|
||||
const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX";
|
||||
const POSTHOG_HOST = "https://us.i.posthog.com/i/v0/e/";
|
||||
const REQUEST_TIMEOUT_MS = 2_000;
|
||||
|
||||
function _loadPluginVersion(): string {
|
||||
// Source context: telemetry.ts sits next to package.json (./).
|
||||
// Bundled context: dist/index.js sits one level below it (../).
|
||||
for (const rel of ["./package.json", "../package.json"]) {
|
||||
try {
|
||||
const pkg = JSON.parse(readFileSync(new URL(rel, import.meta.url), "utf-8"));
|
||||
if (pkg?.name === "@mem0/opencode-plugin" && pkg.version) return pkg.version;
|
||||
} catch {
|
||||
/* try next candidate */
|
||||
}
|
||||
}
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
const PLUGIN_VERSION = _loadPluginVersion();
|
||||
|
||||
export function isTelemetryEnabled(): boolean {
|
||||
const val = process.env.MEM0_TELEMETRY;
|
||||
if (val === undefined) return true;
|
||||
const s = val.toLowerCase();
|
||||
return s !== "false" && s !== "0" && s !== "no" && s !== "off";
|
||||
}
|
||||
|
||||
function distinctId(apiKey: string): string {
|
||||
// Matches telemetry.py `_distinct_id()` so the same user is one person in
|
||||
// PostHog whether they use OpenCode or any other Mem0 editor plugin.
|
||||
return createHash("sha256").update(apiKey).digest("hex").slice(0, 32);
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the PostHog event payload, or null when telemetry is disabled or no
|
||||
* API key is available. Pure (aside from env/version reads) and exported for
|
||||
* testing. System-controlled properties are applied last so a caller cannot
|
||||
* override `source`/`platform`/etc.
|
||||
*/
|
||||
export function buildEvent(
|
||||
eventType: string,
|
||||
properties: Record<string, unknown>,
|
||||
apiKey: string | undefined,
|
||||
): Record<string, unknown> | null {
|
||||
if (!isTelemetryEnabled() || !apiKey) return null;
|
||||
return {
|
||||
api_key: POSTHOG_API_KEY,
|
||||
distinct_id: distinctId(apiKey),
|
||||
event: `plugin.${eventType}`,
|
||||
properties: {
|
||||
...properties,
|
||||
source: "plugin",
|
||||
platform: "opencode",
|
||||
plugin_version: PLUGIN_VERSION,
|
||||
os: process.platform,
|
||||
sample_rate: 1.0,
|
||||
$process_person_profile: false,
|
||||
$lib: "posthog-node",
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/** Send a usage event, fire-and-forget. Never throws, never blocks. */
|
||||
export function captureEvent(
|
||||
eventType: string,
|
||||
properties: Record<string, unknown>,
|
||||
apiKey: string | undefined,
|
||||
): void {
|
||||
const payload = buildEvent(eventType, properties, apiKey);
|
||||
if (!payload) return;
|
||||
try {
|
||||
void fetch(POSTHOG_HOST, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(payload),
|
||||
signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
|
||||
}).catch(() => {
|
||||
/* fire-and-forget */
|
||||
});
|
||||
} catch {
|
||||
/* never throw */
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -16,5 +16,5 @@
|
||||
"emitDeclarationOnly": true
|
||||
},
|
||||
"include": ["**/*"],
|
||||
"exclude": ["node_modules", "dist"]
|
||||
"exclude": ["node_modules", "dist", "**/*.test.ts"]
|
||||
}
|
||||
@@ -2,6 +2,21 @@
|
||||
|
||||
All notable changes to the Mem0 plugin will be documented in this file.
|
||||
|
||||
## 0.2.10 — Accurate per-editor telemetry attribution
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Antigravity counted as Claude Code:** `detect_platform()` (`scripts/telemetry.py`) now checks `ANTIGRAVITY_PLUGIN_ROOT` before the `CLAUDE_PLUGIN_ROOT` branch. Antigravity sets both env vars for compatibility, so every Antigravity session was previously attributed to `claude-code`. Telemetry now reports `platform: "antigravity"`.
|
||||
- **Codex fell back to the generic `plugin` bucket:** Codex installs standalone hooks with absolute paths via `install_codex_hooks.py`, so `PLUGIN_ROOT` is never set at runtime and platform auto-detection failed. Each command in `hooks/codex-hooks.json` now pins `MEM0_PLATFORM=codex` inline (Codex runs hook commands through a shell). Telemetry now reports `platform: "codex"`.
|
||||
- **Cursor attribution depended on the host env:** Cursor's `*_cursor.sh` wrappers delegate to the shared hook scripts, whose platform detection relied on Cursor exporting `CURSOR_PLUGIN_ROOT` to the subprocess. All five Cursor wrappers now `export MEM0_PLATFORM=cursor` before delegating.
|
||||
- **`plugin_version` was identical for every editor:** telemetry read `.claude-plugin/plugin.json` for all bash-hook editors, so Antigravity reported `0.2.10` instead of its real `0.1.2`. `_load_plugin_version()` now reads the manifest matching the detected platform, so each editor reports its own version.
|
||||
|
||||
### Added
|
||||
|
||||
- **`MEM0_PLATFORM` override in `detect_platform()`:** An explicit platform marker that wins over env-var auto-detection, letting each editor label its telemetry reliably. New tests in `tests/test_telemetry.py` cover the override, Antigravity attribution, and the Cursor/Codex platform-pinning contracts.
|
||||
|
||||
> Attribution fixes apply to telemetry emitted after users upgrade to this version; PostHog does not backfill past events.
|
||||
|
||||
## 0.2.9 — File-context injection, session summaries & activity timeline
|
||||
|
||||
### Added
|
||||
@@ -90,14 +90,14 @@ git clone https://github.com/mem0ai/mem0.git ~/codex-plugins/mem0-source
|
||||
codex plugin marketplace add ~/codex-plugins/mem0-source
|
||||
```
|
||||
|
||||
This points Codex at the repo's `.agents/plugins/marketplace.json`, which references `mem0-plugin/` as the local source. Restart Codex, run `/plugins`, and install **Mem0** from the **Mem0 Plugins** marketplace.
|
||||
This points Codex at the repo's `.agents/plugins/marketplace.json`, which references `integrations/mem0-plugin/` as the local source. Restart Codex, run `/plugins`, and install **Mem0** from the **Mem0 Plugins** marketplace.
|
||||
|
||||
> **Don't combine with Option A.** The plugin manifest auto-registers `mem0` as an MCP server via `mem0-plugin/.codex-mcp.json` — adding a manual `[mcp_servers.mem0]` block would duplicate the registration.
|
||||
> **Don't combine with Option A.** The plugin manifest auto-registers `mem0` as an MCP server via `integrations/mem0-plugin/.codex-mcp.json` — adding a manual `[mcp_servers.mem0]` block would duplicate the registration.
|
||||
|
||||
**Optional — enable lifecycle hooks.** Codex doesn't auto-wire hooks from plugin manifests; it only reads `~/.codex/hooks.json` (or `<repo>/.codex/hooks.json`) ([docs](https://developers.openai.com/codex/hooks)). Run the bundled installer once to merge Mem0's entries:
|
||||
|
||||
```bash
|
||||
python3 ~/codex-plugins/mem0-source/mem0-plugin/scripts/install_codex_hooks.py
|
||||
python3 ~/codex-plugins/mem0-source/integrations/mem0-plugin/scripts/install_codex_hooks.py
|
||||
```
|
||||
|
||||
This merges three entries into `~/.codex/hooks.json` with absolute paths pointing into your clone:
|
||||
@@ -190,7 +190,7 @@ See [OpenCode integration docs](https://docs.mem0.ai/integrations/opencode) for
|
||||
|
||||
```bash
|
||||
# Install the plugin (MCP server, hooks, scripts)
|
||||
npx degit mem0ai/mem0/mem0-plugin ~/.gemini/config/plugins/mem0
|
||||
npx degit mem0ai/mem0/integrations/mem0-plugin ~/.gemini/config/plugins/mem0
|
||||
```
|
||||
|
||||
This installs the MCP server, lifecycle hooks, and shared scripts.
|
||||
@@ -276,10 +276,10 @@ The background setup is idempotent and runs once per account (cached in `~/.mem0
|
||||
|
||||
```bash
|
||||
# Dry-run -- prints current vs proposed, no changes:
|
||||
python mem0-plugin/scripts/setup_coding_categories.py
|
||||
python integrations/mem0-plugin/scripts/setup_coding_categories.py
|
||||
|
||||
# Write explicitly:
|
||||
python mem0-plugin/scripts/setup_coding_categories.py --apply
|
||||
python integrations/mem0-plugin/scripts/setup_coding_categories.py --apply
|
||||
```
|
||||
|
||||
Requires the `mem0ai` Python SDK (`pip install mem0ai`) and `MEM0_API_KEY` set. `project.update(custom_categories=[...])` always replaces the full list.
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user