Compare commits
7 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 19f7134082 | |||
| a8d3634312 | |||
| 1a5c7ad28c | |||
| 4e38b057fd | |||
| 3362999095 | |||
| 012cd32c3a | |||
| e4e0307ae6 |
@@ -12,7 +12,7 @@
|
||||
"name": "mem0",
|
||||
"source": "./integrations/claude-code-plugin",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"version": "0.3.1"
|
||||
"version": "0.3.2"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
"name": "mem0",
|
||||
"source": "./integrations/cursor-plugin",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"version": "0.3.1"
|
||||
"version": "0.3.2"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -32,6 +32,9 @@ jobs:
|
||||
- name: Type check
|
||||
run: bun run type-check
|
||||
|
||||
- name: Test
|
||||
run: bun test
|
||||
|
||||
- name: Build
|
||||
run: bun run build
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
{
|
||||
"id": "mem0",
|
||||
"displayName": "Mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"homepage": "https://mem0.ai",
|
||||
"keywords": ["memory", "personalization", "mcp", "semantic-search"],
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@mem0/cli",
|
||||
"version": "0.2.13",
|
||||
"version": "0.2.14",
|
||||
"description": "The official CLI for mem0 — the memory layer for AI agents",
|
||||
"type": "module",
|
||||
"bin": {
|
||||
|
||||
@@ -31,7 +31,8 @@ export class PlatformBackend implements Backend {
|
||||
this.headers = {
|
||||
Authorization: `Token ${config.apiKey}`,
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": "cli",
|
||||
"X-Mem0-Source": "CLI",
|
||||
"X-Mem0-Client": `mem0-cli-node/${CLI_VERSION}`,
|
||||
"X-Mem0-Client-Language": "node",
|
||||
"X-Mem0-Client-Version": CLI_VERSION,
|
||||
};
|
||||
|
||||
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "mem0-cli"
|
||||
version = "0.2.12"
|
||||
version = "0.2.13"
|
||||
description = "The official CLI for mem0 — the memory layer for AI agents"
|
||||
readme = "README.md"
|
||||
license = "Apache-2.0"
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
"""mem0 CLI — the command-line interface for the mem0 memory layer."""
|
||||
|
||||
__version__ = "0.2.12"
|
||||
__version__ = "0.2.13"
|
||||
|
||||
@@ -27,7 +27,8 @@ class PlatformBackend(Backend):
|
||||
headers={
|
||||
"Authorization": f"Token {config.api_key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": "cli",
|
||||
"X-Mem0-Source": "CLI",
|
||||
"X-Mem0-Client": f"mem0-cli-python/{__version__}",
|
||||
"X-Mem0-Client-Language": "python",
|
||||
"X-Mem0-Client-Version": __version__,
|
||||
},
|
||||
|
||||
+96
-35
@@ -7,6 +7,14 @@ mode: "wide"
|
||||
<Tabs>
|
||||
<Tab title="Python">
|
||||
|
||||
<Update label="2026-09-18" description="v2.1.0">
|
||||
|
||||
**Improvements:**
|
||||
- **Client:** Requests now carry three surface-identity headers so the platform can tell which product made a call. `X-Mem0-Source` names the surface and `X-Application` the host app it runs inside, both set-once so a wrapper that already declared its identity keeps it. `X-Mem0-Client` is append-only and carries `name/version` per layer, outermost first, so a plugin calling this SDK reports the whole chain rather than only the last speaker. `MEM0_SOURCE`, `MEM0_APPLICATION` and `MEM0_CLIENT_STACK` set them from the environment for wrappers that cannot pass options ([#7326](https://github.com/mem0ai/mem0/pull/7326))
|
||||
- **Client:** The client stack is bounded by dropping whole entries rather than slicing characters, and this SDK's own entry is the reserved one. Truncating the joined string could sever an identifier mid-name and the platform parsed the fragment as a real client ([#7326](https://github.com/mem0ai/mem0/pull/7326))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-09-02" description="v2.0.20">
|
||||
|
||||
**Improvements:**
|
||||
@@ -1227,6 +1235,14 @@ See the [OSS v2 to v3 migration guide](https://docs.mem0.ai/migration/oss-v2-to-
|
||||
|
||||
<Tab title="TypeScript">
|
||||
|
||||
<Update label="2026-09-18" description="v3.2.0">
|
||||
|
||||
**Improvements:**
|
||||
- **Client:** Requests now carry `X-Mem0-Source`, `X-Application` and `X-Mem0-Client`, matching the Python SDK. The first two are set-once so an outer wrapper keeps its identity; the third is append-only and reports the whole layer chain. Read from `MEM0_SOURCE`, `MEM0_APPLICATION` and `MEM0_CLIENT_STACK` when set ([#7326](https://github.com/mem0ai/mem0/pull/7326))
|
||||
- **Client:** The SDK version in `X-Mem0-Client` is injected at build time rather than hardcoded, so it cannot go stale at the next release ([#7326](https://github.com/mem0ai/mem0/pull/7326))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-09-02" description="v3.1.8">
|
||||
|
||||
**Improvements:**
|
||||
@@ -1864,6 +1880,13 @@ See the [OSS v2 to v3 migration guide](https://docs.mem0.ai/migration/oss-v2-to-
|
||||
|
||||
<Tab title="CLI">
|
||||
|
||||
<Update label="2026-09-18" description="Python v0.2.13 / Node v0.2.14">
|
||||
|
||||
**Improvements:**
|
||||
- **Client:** Requests now carry the three surface-identity headers (`X-Mem0-Source`, `X-Application`, `X-Mem0-Client`) introduced in the Python and TypeScript SDKs, so the platform can attribute calls made through the CLI to the correct surface and version ([#7326](https://github.com/mem0ai/mem0/pull/7326))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-08-24" description="Python v0.2.12 / Node v0.2.13">
|
||||
|
||||
**New Features:**
|
||||
@@ -2076,7 +2099,7 @@ A full-featured command-line interface for Mem0, available in both Python and No
|
||||
- New Git repository writes use a hash of the remote identity in `agent_id`. Search and explicit shared-memory deletion include both current and legacy repository IDs within the repository's `app_id`. Existing memories are not rewritten. Legacy IDs retain their original ambiguity for matching owner/repository names on different Git hosts.
|
||||
|
||||
**Packaging:**
|
||||
- Claude Code, Cursor, Codex, Kimi, Antigravity, and the portable Python bundle are versioned at `0.3.1`. OpenCode, Pi Agent, and DeepSeek Harness are `0.3.0`; OpenClaw is `1.1.0`. Each host's changes and upgrade considerations are listed in its tab.
|
||||
- Claude Code, Cursor, Codex, Kimi, Antigravity, and the portable Python bundle are versioned at `0.3.1`. OpenCode and DeepSeek Harness are `0.3.0`; Pi Agent is `0.3.0`; OpenClaw is `1.1.0`. Each host's changes and upgrade considerations are listed in its tab.
|
||||
- Python and TypeScript CI run their respective runtime suites. Package checks build the installable artifacts, check generated-file consistency, and reject TypeScript output that still imports monorepo source.
|
||||
|
||||
[#7203](https://github.com/mem0ai/mem0/pull/7203)
|
||||
@@ -2371,9 +2394,14 @@ Initial release of the Mem0 plugin for Claude Code and Cursor, followed by Codex
|
||||
|
||||
<Tab title="Claude Code">
|
||||
|
||||
<Update label="Unreleased" description="Sidekick availability">
|
||||
<Update label="2026-09-18" description="Claude Code plugin v0.3.2">
|
||||
|
||||
Sidekick is now available only in Claude Code, with Sonnet, worktree isolation, and parent memories.
|
||||
**Improvements:**
|
||||
- **Telemetry:** `PLUGIN_VERSION` bumped to `0.3.2`. The `mem0-plugin/<version>` wire header and `plugin_version` telemetry field now reflect the fixes from #7322 through #7358 ([#7373](https://github.com/mem0ai/mem0/pull/7373))
|
||||
- **Telemetry:** Events are no longer delivered twice, no longer lose parked events on flush, and now attribute each event to the plugin that produced it ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
**Changes:**
|
||||
- **Sidekick:** Sidekick is now available only in Claude Code, with Sonnet, worktree isolation, and parent memories.
|
||||
|
||||
</Update>
|
||||
|
||||
@@ -2396,9 +2424,14 @@ Sidekick is now available only in Claude Code, with Sonnet, worktree isolation,
|
||||
|
||||
<Tab title="Cursor">
|
||||
|
||||
<Update label="Unreleased" description="Sidekick availability">
|
||||
<Update label="2026-09-18" description="Cursor plugin v0.3.2">
|
||||
|
||||
Removes Sidekick and its start/stop hooks. Memory capture, search, and six skills remain available.
|
||||
**Improvements:**
|
||||
- **Telemetry:** `PLUGIN_VERSION` bumped to `0.3.2`. The `mem0-plugin/<version>` wire header and `plugin_version` telemetry field now reflect the fixes from #7322 through #7358 ([#7373](https://github.com/mem0ai/mem0/pull/7373))
|
||||
- **Telemetry:** Events are no longer delivered twice, no longer lose parked events on flush, and now attribute each event to the plugin that produced it ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
**Changes:**
|
||||
- **Sidekick:** Removes Sidekick and its start/stop hooks. Memory capture, search, and six skills remain available.
|
||||
|
||||
</Update>
|
||||
|
||||
@@ -2421,9 +2454,14 @@ Removes Sidekick and its start/stop hooks. Memory capture, search, and six skill
|
||||
|
||||
<Tab title="Codex">
|
||||
|
||||
<Update label="Unreleased" description="Sidekick availability">
|
||||
<Update label="2026-09-18" description="Codex plugin v0.3.2">
|
||||
|
||||
Renames shared tracking to use subagent terminology. Native subagent memory support remains available.
|
||||
**Improvements:**
|
||||
- **Telemetry:** `PLUGIN_VERSION` bumped to `0.3.2`. The `mem0-plugin/<version>` wire header and `plugin_version` telemetry field now reflect the fixes from #7322 through #7358 ([#7373](https://github.com/mem0ai/mem0/pull/7373))
|
||||
- **Telemetry:** Events are no longer delivered twice, no longer lose parked events on flush, and now attribute each event to the plugin that produced it ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
**Changes:**
|
||||
- **Sidekick:** Renames shared tracking to use subagent terminology. Native subagent memory support remains available.
|
||||
|
||||
</Update>
|
||||
|
||||
@@ -2443,32 +2481,16 @@ Renames shared tracking to use subagent terminology. Native subagent memory supp
|
||||
|
||||
</Tab>
|
||||
|
||||
<Tab title="Agent Plugins v1">
|
||||
|
||||
<Update label="Unreleased" description="Sidekick availability">
|
||||
|
||||
Sidekick is available only in Claude Code, not in the portable package.
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-09-08" description="Portable Mem0 plugin v0.3.1">
|
||||
|
||||
**Added:**
|
||||
- One portable package at `integrations/mem0-agent-plugin/`, using the Agent Plugins 1.0.0 root `plugin.json`, `mcp.json`, and fixed `skills/` locations.
|
||||
- Ships a local, read-only `search_memories` server and the six shared memory skills. Uses `PLUGIN_ROOT` for bundled files and `PLUGIN_DATA` for persistent plugin state; all package files remain inside the installable directory.
|
||||
|
||||
**Packaging:**
|
||||
- Generated from the shared Python runtime and skill templates. Builds validate the manifest, MCP configuration, skills, and generated-file consistency.
|
||||
- Host lifecycle hooks and native Sidekick declarations remain in the native plugin packages; the portable package does not provide automatic lifecycle capture or host-specific subagent isolation. Its bundled remember skill cannot persist a new memory on its own because the portable package has no capture hooks or write tool.
|
||||
|
||||
[#7203](https://github.com/mem0ai/mem0/pull/7203)
|
||||
|
||||
</Update>
|
||||
|
||||
</Tab>
|
||||
|
||||
<Tab title="OpenCode">
|
||||
|
||||
<Update label="2026-09-18" description="OpenCode plugin v0.4.0">
|
||||
|
||||
**Changes:**
|
||||
- **Telemetry:** The PostHog `source` tag changed from the literal `"plugin"` to `OPENCODE_PLUGIN`, and `project_hash` is now salted. Saved PostHog insights filtering on `source = "plugin"` will stop matching new events; historical data is unaffected ([#7322](https://github.com/mem0ai/mem0/pull/7322))
|
||||
- **Config:** A new `keyFingerprint` key appears in the install-count deduplication logic; installs are now counted once per key rather than on every activation ([#7325](https://github.com/mem0ai/mem0/pull/7325))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-09-08" description="OpenCode plugin v0.3.0">
|
||||
|
||||
**Changed:**
|
||||
@@ -2567,9 +2589,14 @@ Sidekick is available only in Claude Code, not in the portable package.
|
||||
|
||||
<Tab title="Antigravity">
|
||||
|
||||
<Update label="Unreleased" description="Sidekick availability">
|
||||
<Update label="2026-09-18" description="Antigravity plugin v0.3.2">
|
||||
|
||||
Removes Sidekick. Memory capture, search, and six skills remain available.
|
||||
**Improvements:**
|
||||
- **Telemetry:** `PLUGIN_VERSION` bumped to `0.3.2`. The `mem0-plugin/<version>` wire header and `plugin_version` telemetry field now reflect the fixes from #7322 through #7358 ([#7373](https://github.com/mem0ai/mem0/pull/7373))
|
||||
- **Telemetry:** Events are no longer delivered twice, no longer lose parked events on flush, and now attribute each event to the plugin that produced it ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
**Changes:**
|
||||
- **Sidekick:** Removes Sidekick. Memory capture, search, and six skills remain available.
|
||||
|
||||
</Update>
|
||||
|
||||
@@ -2665,9 +2692,14 @@ Existing memories written by the previous versions are not rewritten. If your me
|
||||
|
||||
<Tab title="Kimi">
|
||||
|
||||
<Update label="Unreleased" description="Sidekick availability">
|
||||
<Update label="2026-09-18" description="Kimi Code plugin v0.3.2">
|
||||
|
||||
Removes Sidekick and its start/stop hooks. Memory capture, recall, and six skills remain available.
|
||||
**Improvements:**
|
||||
- **Telemetry:** `PLUGIN_VERSION` bumped to `0.3.2`. The `mem0-plugin/<version>` wire header and `plugin_version` telemetry field now reflect the fixes from #7322 through #7358 ([#7373](https://github.com/mem0ai/mem0/pull/7373))
|
||||
- **Telemetry:** Events are no longer delivered twice, no longer lose parked events on flush, and now attribute each event to the plugin that produced it ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
**Changes:**
|
||||
- **Sidekick:** Removes Sidekick and its start/stop hooks. Memory capture, recall, and six skills remain available.
|
||||
|
||||
</Update>
|
||||
|
||||
@@ -2705,6 +2737,14 @@ Removes Sidekick and its start/stop hooks. Memory capture, recall, and six skill
|
||||
|
||||
<Tab title="OpenClaw">
|
||||
|
||||
<Update label="2026-09-18" description="openclaw-mem0 v1.2.0">
|
||||
|
||||
**Changes:**
|
||||
- **Config:** Added `keyFingerprint` to the config schema for install-count deduplication; installs are now counted once per key rather than on every activation ([#7325](https://github.com/mem0ai/mem0/pull/7325))
|
||||
- **Telemetry:** Events are no longer delivered twice, and the `plugin_version` field now reflects the plugin that produced the event ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-09-08" description="openclaw-mem0 v1.1.0">
|
||||
|
||||
**Changed:**
|
||||
@@ -2993,6 +3033,13 @@ Removes Sidekick and its start/stop hooks. Memory capture, recall, and six skill
|
||||
|
||||
<Tab title="Pi Agent">
|
||||
|
||||
<Update label="2026-09-18" description="Pi Agent plugin v0.3.1">
|
||||
|
||||
**Improvements:**
|
||||
- **Telemetry:** Events are no longer delivered twice, no longer lose parked events, and now attribute each event to the plugin that produced it. The `plugin_version` wire field reflects the fixed release ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-09-08" description="Pi Agent plugin v0.3.0">
|
||||
|
||||
**Changed:**
|
||||
@@ -3106,6 +3153,13 @@ Removes Sidekick and its start/stop hooks. Memory capture, recall, and six skill
|
||||
|
||||
<Tab title="DeepSeek Harness">
|
||||
|
||||
<Update label="2026-09-18" description="deepseek-plugin v0.3.1">
|
||||
|
||||
**Improvements:**
|
||||
- **Telemetry:** Rebuild with the fixed shared telemetry core from `agent-plugin-core`. Events are no longer delivered twice, no longer lose parked events on flush, and now attribute each event to the plugin that produced it ([#7323](https://github.com/mem0ai/mem0/pull/7323), [#7324](https://github.com/mem0ai/mem0/pull/7324), [#7358](https://github.com/mem0ai/mem0/pull/7358))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-09-08" description="deepseek-plugin v0.3.0">
|
||||
|
||||
**Added:**
|
||||
@@ -3150,6 +3204,13 @@ Removes Sidekick and its start/stop hooks. Memory capture, recall, and six skill
|
||||
|
||||
<Tab title="Vercel AI SDK">
|
||||
|
||||
<Update label="2026-09-18" description="Vercel AI SDK v3.0.3">
|
||||
|
||||
**Improvements:**
|
||||
- **Client:** Inherits the three surface-identity headers (`X-Mem0-Source`, `X-Application`, `X-Mem0-Client`) from the TypeScript SDK bump, so platform calls made through the Vercel AI SDK provider are now correctly attributed ([#7326](https://github.com/mem0ai/mem0/pull/7326))
|
||||
|
||||
</Update>
|
||||
|
||||
<Update label="2026-08-24" description="Vercel AI SDK v3.0.2">
|
||||
|
||||
**Security:**
|
||||
|
||||
@@ -172,6 +172,36 @@ claude plugin update mem0@mem0-plugins --scope user
|
||||
| Sidekick won't start | Must be in a Git repo. Check that your Claude Code version supports plugin agents and worktrees. |
|
||||
| Remove the plugin | `claude plugin uninstall mem0@mem0-plugins` |
|
||||
|
||||
## Telemetry
|
||||
|
||||
The plugin sends usage events (which hook ran, timing, result counts, failure
|
||||
types) so Mem0 can see what's used and what's breaking.
|
||||
|
||||
These events are **not anonymous**. When an API key is configured, which
|
||||
installing the plugin requires, they are sent under your Mem0 account email,
|
||||
the same way the Python SDK and the CLI attribute theirs. Without a key they
|
||||
are sent under a random per-machine id.
|
||||
|
||||
Each event carries the event name, the plugin version, the harness it ran in,
|
||||
your OS and Python version, and per-event properties describing what happened:
|
||||
timings, counts, coarse outcome and failure labels, and which model was
|
||||
configured. Repository and session identifiers are hashed with a random salt
|
||||
generated on your machine, so they cannot be linked back to a repository name
|
||||
or path.
|
||||
|
||||
The exact set is enforced in code rather than by this list: every property is
|
||||
filtered through a denylist of sensitive keys and credential-shaped values are
|
||||
redacted before anything is sent.
|
||||
|
||||
Prompts, memory text, queries, file paths, repository names, and API keys are
|
||||
never sent.
|
||||
|
||||
Turn it off:
|
||||
|
||||
```bash
|
||||
export MEM0_TELEMETRY=false
|
||||
```
|
||||
|
||||
<CardGroup cols={2}>
|
||||
<Card title="Mem0 MCP Setup" icon="puzzle-piece" href="/platform/mem0-mcp">
|
||||
Detailed MCP configuration for all clients
|
||||
|
||||
@@ -114,7 +114,7 @@ Both tools also accept optional per-call `userId`, `agentId`, and `runId` params
|
||||
|
||||
## Telemetry
|
||||
|
||||
Writes are tagged `source="DEEPSEEK_HARNESS"` so Mem0 can attribute usage to this integration. Anonymous usage events include operation names, durations, result counts, and coarse failure kinds. Queries, memory text, entity IDs, and API keys are never included. Set `MEM0_TELEMETRY=false` to opt out.
|
||||
Writes are tagged `source="DEEPSEEK_HARNESS"` so Mem0 can attribute usage to this integration. Usage events include operation names, durations, result counts, and coarse failure kinds. They are **not anonymous**: when an API key is configured they are sent under your Mem0 account email, the same way the SDK attributes its own. Queries, memory text, entity IDs, and API keys are never included. Set `MEM0_TELEMETRY=false` to opt out.
|
||||
|
||||
<Note>
|
||||
This plugin is a developer preview and tracks the evolving DeepSeek Harness plugin API.
|
||||
|
||||
@@ -481,7 +481,9 @@ Plugin config is stored in `~/.openclaw/openclaw.json` with file permissions `0o
|
||||
|
||||
### Telemetry
|
||||
|
||||
Anonymous usage telemetry (PostHog) is enabled by default to help improve the plugin. No conversation content or memory values are included, only event counts (recall, capture, tool usage, CLI commands).
|
||||
Usage telemetry (PostHog) is enabled by default to help improve the plugin. No conversation content or memory values are included, only event counts (recall, capture, tool usage, CLI commands).
|
||||
|
||||
These events are **not anonymous**. OpenClaw does not send your account email the way the SDK does, but it does send an unsalted SHA-256 hash of it, falling back to a hash of the API key and then to a random per-machine id. Mem0 holds the email the hash is derived from, so the hash identifies your account rather than concealing it. The first run that resolves an account also emits a PostHog `$identify`, which permanently merges any earlier random id into that identity.
|
||||
|
||||
To opt out, set the environment variable:
|
||||
|
||||
|
||||
@@ -49,6 +49,48 @@ Run the type check after every TypeScript change: `pnpm run typecheck` or `tsc -
|
||||
- **`zapier-mem0/`** is a Zapier Platform CLI app: add, search, get, delete. It deploys to Zapier, not npm, so it is **not** in the release router. Deploy it with `gh workflow run zapier-mem0-cd.yml --ref main` (needs the `ZAPIER_DEPLOY_KEY` secret).
|
||||
- **`mem0-strands/`** is a native Strands `MemoryStore` (Python, published to PyPI as `mem0-strands`). It plugs into the Strands `MemoryManager` for automatic recall and server-side extraction, over the hosted Mem0 platform or self-hosted Mem0 OSS. The package lives under `mem0-strands/python/`.
|
||||
|
||||
## Surface attribution
|
||||
|
||||
Every integration tells the Mem0 platform which surface it is. Three headers,
|
||||
and the rules on them are what keep one layer from erasing another:
|
||||
|
||||
| Header | Carries | Rule |
|
||||
|--------|---------|------|
|
||||
| `X-Mem0-Source` | one canonical source value | **set-once** — write only if absent |
|
||||
| `X-Application` | the host app it runs inside | **set-once** — write only if absent |
|
||||
| `X-Mem0-Client` | `name/version`, outermost first | **append-only** — add yourself, never replace |
|
||||
|
||||
Set-once means check-then-set, never assignment. An integration that wraps the
|
||||
SDK is the outermost layer and sets the source; the SDK underneath defers to it.
|
||||
Assignment is exactly how every agent plugin came to be indistinguishable from
|
||||
every other one at the platform.
|
||||
|
||||
How to declare it from an integration, in order of preference:
|
||||
|
||||
1. Send the headers yourself, if you make the HTTP call directly.
|
||||
2. Pass `source` in the call options, if you go through an SDK.
|
||||
3. Set `MEM0_SOURCE` / `MEM0_APPLICATION` / `MEM0_CLIENT_STACK` in the
|
||||
environment before constructing the client. The SDKs read these and defer to
|
||||
anything already present.
|
||||
|
||||
Append-only applies where a stack can actually form: an SDK handed a client that
|
||||
already carries `X-Mem0-Client` appends itself rather than replacing. An SDK
|
||||
constructed with no outer context simply reports itself, which is correct — it
|
||||
is the outermost layer in that process.
|
||||
|
||||
The backend recognizes a fixed list of source values and buckets everything else
|
||||
into `OTHERS`. A new value has to land in the platform's `EventSource` enum, so
|
||||
do not invent one without that change going in too.
|
||||
|
||||
`X-Application` is allowlisted the same way, and this one has a rule of its own:
|
||||
**omit the header when you do not know the host.** A value outside the allowlist
|
||||
is discarded server-side, so guessing produces an event that claims an
|
||||
attribution we do not actually have. The portable bundle is the case that
|
||||
matters. It runs in whatever editor a user drops it into, so its build leaves
|
||||
`PLATFORM_APPLICATION` empty and `memory_core` sends no header at all, while the
|
||||
native bundles each name the host they were generated for. If you add a build
|
||||
target, decide which of those two it is.
|
||||
|
||||
## Adding an integration
|
||||
|
||||
1. For a native coding-agent host, add `integrations/<name>-plugin/` with `plugin-build.json`, its manifest, and a thin adapter, then generate its shared runtime. Portable clients use the single `mem0-agent-plugin/` package. Independent TypeScript integrations stay self-contained and import shared lifecycle behavior from `agent-plugin-core/typescript/`.
|
||||
@@ -59,3 +101,4 @@ Run the type check after every TypeScript change: `pnpm run typecheck` or `tsc -
|
||||
5. If it is a Claude Code or editor marketplace plugin, register the generated native bundle path in the applicable marketplace files. Preserve the existing public plugin name.
|
||||
6. Document it under `docs/integrations/` and add the page to `docs/docs.json` and `docs/llms.txt`.
|
||||
7. Add rows to the table above and to the CI/CD tables in [`../.github/AGENTS.md`](../.github/AGENTS.md).
|
||||
8. Send the three headers in [Surface attribution](#surface-attribution), and land the matching `EventSource` value on the platform in the same week. Until it exists, your traffic reports as `OTHERS`.
|
||||
|
||||
@@ -81,6 +81,38 @@ def replace_output(staged: Path, output: Path) -> Path:
|
||||
return output
|
||||
|
||||
|
||||
def _render_harness_id(host: str, *, portable: bool = False) -> str:
|
||||
"""Emit core/_harness_id.py for one host.
|
||||
|
||||
Carries both vocabularies from a single definition: the PostHog `source` tag
|
||||
and the platform's X-Mem0-Source / X-Application pair. Keeping them together
|
||||
is what stops the two from drifting into separate vocabularies for the same
|
||||
thing.
|
||||
|
||||
The portable bundle runs in whatever editor a user drops it into, so it does
|
||||
not know its host and must not guess one. HARNESS_ID stays "coding-agent",
|
||||
which is true and useful for grouping in PostHog, but PLATFORM_APPLICATION is
|
||||
left empty: X-Application names a real host app, is checked against an
|
||||
allowlist server-side, and a value that is always discarded is worse than no
|
||||
value -- it reads like an attribution we have and do not.
|
||||
"""
|
||||
tag = host.upper().replace("-", "_") + "_PLUGIN"
|
||||
application = "" if portable else host
|
||||
return (
|
||||
'"""Generated by integrations/agent-plugin-core/build/build.py. Do not edit."""\n'
|
||||
"\n"
|
||||
f'HARNESS_ID = "{host}"\n'
|
||||
f'SOURCE_TAG = "{tag}"\n'
|
||||
"\n"
|
||||
"# Platform-side vocabulary (mem0_event.source + X-Application). The whole\n"
|
||||
"# plugin family is one source; which editor it runs in is the application.\n"
|
||||
"# An empty application means the host is unknown, and memory_core omits\n"
|
||||
"# the header entirely rather than sending a placeholder.\n"
|
||||
'PLATFORM_SOURCE = "MEM0_PLUGIN"\n'
|
||||
f'PLATFORM_APPLICATION = "{application}"\n'
|
||||
)
|
||||
|
||||
|
||||
def _bundle_python(
|
||||
staged: Path,
|
||||
host: str,
|
||||
@@ -96,6 +128,12 @@ def _bundle_python(
|
||||
continue
|
||||
shutil.copy2(source, core / source.name)
|
||||
|
||||
# Generated per host so identity does not depend on an entrypoint remembering
|
||||
# to call telemetry.init(). mcp_server.py and the detached telemetry.py sender
|
||||
# never did, which is how MCP searches reported harness=generic and every
|
||||
# batch they drained was labelled MEM0_PLUGIN regardless of the real host.
|
||||
(core / "_harness_id.py").write_text(_render_harness_id(host, portable=portable), encoding="utf-8")
|
||||
|
||||
values = {
|
||||
"PLUGIN_ROOT": plugin_root,
|
||||
"PLUGIN_DATA": "${PLUGIN_DATA}",
|
||||
|
||||
@@ -290,6 +290,11 @@ def run(
|
||||
if args.plugin_data_dir:
|
||||
os.environ[data_dir_env] = args.plugin_data_dir
|
||||
|
||||
# Snapshot BEFORE anything writes to the data dir: cache_plugin_api_key
|
||||
# writes `api-key` and EvidenceStore creates `evidence.sqlite3`, so asking
|
||||
# after them always saw content and every fresh install reported an upgrade.
|
||||
data_dir_was_empty = telemetry.data_dir_was_empty()
|
||||
|
||||
cache_plugin_api_key()
|
||||
if args.action == "session-start":
|
||||
clear_stale_api_key_cache()
|
||||
@@ -305,8 +310,19 @@ def run(
|
||||
return 0
|
||||
|
||||
if args.action == "session-start":
|
||||
if telemetry.is_first_run():
|
||||
# Claims the marker atomically and says which event to record, so a
|
||||
# second session starting alongside this one cannot record it too.
|
||||
first_event = telemetry.claim_install(was_empty=data_dir_was_empty)
|
||||
if first_event == "install":
|
||||
telemetry.record("install")
|
||||
elif first_event == "upgrade":
|
||||
# First run after a build that never wrote the marker; the
|
||||
# predecessor version was never recorded anywhere.
|
||||
telemetry.record("upgrade", from_version="pre-0.3")
|
||||
else:
|
||||
previous = telemetry.claim_version_change()
|
||||
if previous:
|
||||
telemetry.record("upgrade", from_version=previous)
|
||||
recovered = recover_pending_handoffs()
|
||||
record_session_start(store, hook_input)
|
||||
if recovered:
|
||||
|
||||
@@ -29,7 +29,7 @@ from typing import Any, Iterable
|
||||
import telemetry
|
||||
|
||||
DEFAULT_API_URL = "https://api.mem0.ai"
|
||||
PLUGIN_VERSION = "0.3.1"
|
||||
PLUGIN_VERSION = "0.3.2"
|
||||
|
||||
_harness_name: str = "generic"
|
||||
_harness_env_prefix: str = "MEM0_PLUGIN"
|
||||
@@ -1800,6 +1800,34 @@ def extraction_message_batches(
|
||||
return batches
|
||||
|
||||
|
||||
# Platform surface attribution. Read from the generated per-host module so a new
|
||||
# entrypoint is correct without remembering to configure anything.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
except ImportError:
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
|
||||
def platform_headers(key: str) -> dict[str, str]:
|
||||
"""Auth plus the three surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are set-once by contract: this is the
|
||||
outermost layer, so it sets them, and nothing below may overwrite them.
|
||||
X-Mem0-Client is append-only — anything downstream adds itself to the tail.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": _PLATFORM_SOURCE,
|
||||
"X-Mem0-Client": f"mem0-plugin/{PLUGIN_VERSION}",
|
||||
}
|
||||
if _PLATFORM_APPLICATION:
|
||||
headers["X-Application"] = _PLATFORM_APPLICATION
|
||||
return headers
|
||||
|
||||
|
||||
def _request_json(
|
||||
url: str, key: str, payload: dict[str, Any], timeout: float
|
||||
) -> tuple[dict[str, Any] | list[Any], int, int]:
|
||||
@@ -1807,7 +1835,7 @@ def _request_json(
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1834,7 +1862,7 @@ def _get_json(
|
||||
) -> tuple[dict[str, Any] | list[Any], int]:
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="GET",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1980,6 +2008,13 @@ def flush_session(
|
||||
"user_id": write_user,
|
||||
"app_id": repo.app_id,
|
||||
"run_id": session_id,
|
||||
# Top level, not metadata: the backend reads `source` from the body or
|
||||
# the query string, never from metadata, which is where this used to
|
||||
# sit. The X-Mem0-Source header is also read, but only from the
|
||||
# platform release that ships alongside this change, so the body value
|
||||
# is what makes attribution work on both. The harness tag stays in
|
||||
# metadata as hook provenance.
|
||||
"source": _PLATFORM_SOURCE,
|
||||
"metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)},
|
||||
"agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS,
|
||||
"custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS,
|
||||
@@ -2523,7 +2558,7 @@ def _collect_memory_ids(
|
||||
def _delete_memory(api_url: str, key: str, memory_id: str) -> bool:
|
||||
request = urllib.request.Request(
|
||||
f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/",
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Anonymous usage telemetry for Mem0 agent plugins.
|
||||
"""Usage telemetry for Mem0 agent plugins.
|
||||
|
||||
Events are linked to your Mem0 account email when an API key is configured, and
|
||||
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
||||
attribute the same way.
|
||||
|
||||
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
||||
touches the network: `record` appends one JSON line to a local spool and returns.
|
||||
@@ -9,7 +13,8 @@ started once per session and again from the flush worker that is already detache
|
||||
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
||||
|
||||
Never sends prompts, memory text, queries, file paths, repository names, or API
|
||||
keys: only event names, durations, counts, coarse outcomes, and salted hashes.
|
||||
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
||||
identifiers hashed with a random per-install salt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,8 +34,24 @@ from typing import Any
|
||||
|
||||
import memory_core
|
||||
|
||||
_harness: str = "generic"
|
||||
_source_tag: str = "MEM0_PLUGIN"
|
||||
# Seeded from the per-host module the build generates into core/. Two processes
|
||||
# in this pipeline never call init() — mcp_server.py, and the detached
|
||||
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
||||
# was what every one of their events got labelled with.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
||||
except ImportError:
|
||||
_DEFAULT_HARNESS = "generic"
|
||||
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
_salt_cache: str = ""
|
||||
_harness: str = _DEFAULT_HARNESS
|
||||
_source_tag: str = _DEFAULT_SOURCE_TAG
|
||||
_PRIVATE_KEYS = {
|
||||
"apikey",
|
||||
"authorization",
|
||||
@@ -56,10 +77,19 @@ _PRIVATE_KEYS = {
|
||||
}
|
||||
|
||||
|
||||
def init(harness: str = "generic", source_tag: str = "") -> None:
|
||||
def init(harness: str = "", source_tag: str = "") -> None:
|
||||
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
||||
|
||||
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
||||
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
||||
meant one plugin could emit three different source values depending on which
|
||||
process happened to send the batch.
|
||||
"""
|
||||
global _harness, _source_tag
|
||||
_harness = harness
|
||||
_source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN"
|
||||
_harness = harness or _DEFAULT_HARNESS
|
||||
_source_tag = source_tag or (
|
||||
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
||||
)
|
||||
|
||||
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
||||
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
||||
@@ -70,6 +100,16 @@ BATCH_SIZE = 100
|
||||
SEND_TIMEOUT = 5
|
||||
CLAIM_STALE_SECONDS = 120
|
||||
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
||||
# A batch is only discarded once it has genuinely been retried this many times.
|
||||
MAX_CLAIM_ATTEMPTS = 3
|
||||
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
||||
# cannot turn one flush into an unbounded send loop.
|
||||
MAX_PARKED_PER_RUN = 3
|
||||
# Added to the wait before a released claim becomes reclaimable, per attempt
|
||||
# already spent. Releasing straight to "reclaimable now" let two senders burn the
|
||||
# whole budget within seconds of one another on a single momentary failure, and
|
||||
# discard a batch a retry a minute later would have delivered.
|
||||
RETRY_COOLDOWN_SECONDS = 60
|
||||
|
||||
|
||||
def is_enabled() -> bool:
|
||||
@@ -83,9 +123,126 @@ def is_enabled() -> bool:
|
||||
|
||||
|
||||
def _digest(value: str, length: int = 16) -> str:
|
||||
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _salt_path() -> Path:
|
||||
return memory_core.data_dir() / "telemetry-salt"
|
||||
|
||||
|
||||
def _install_salt() -> str:
|
||||
"""Random per-install salt, created once and memoized for the process.
|
||||
|
||||
Deliberately its own file, claimed with O_CREAT|O_EXCL, rather than a key in
|
||||
the identity file. Three reasons, all of which produced wrong data when this
|
||||
lived in the identity dict:
|
||||
|
||||
- Hooks are short-lived separate processes firing on every tool call, and
|
||||
people run more than one agent window. A read-modify-write would let each
|
||||
process mint its own salt, so one repository would hash several ways in the
|
||||
window before a writer won.
|
||||
- resolve_distinct_id holds a copy of the identity dict across a network call
|
||||
to /v1/ping/, so whichever write landed second erased the other's key —
|
||||
losing either the salt (repo_hash changes mid-stream) or the email (a
|
||||
second $identify, splitting the person).
|
||||
- Touching the identity file from record() would create it, and is_first_run
|
||||
keys off that file, so recording an event would silently suppress the
|
||||
install event.
|
||||
|
||||
Published atomically, and there is deliberately no derived fallback. Creating
|
||||
the file with O_CREAT|O_EXCL and then writing into it leaves a window where
|
||||
the file exists and is empty, and a concurrent hook that reads it in that
|
||||
window gets nothing. Falling back to a digest of the path would hand that
|
||||
process a salt an attacker can compute, memoized for its whole run, which is
|
||||
the privacy control this function exists to provide silently turning itself
|
||||
off under load. The salt is written to a private temp file first and linked
|
||||
into place, so the name either does not exist or already has the full value.
|
||||
|
||||
Returns "" when it genuinely cannot persist. Callers omit the hash entirely
|
||||
rather than emit an unsalted one.
|
||||
"""
|
||||
global _salt_cache
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
|
||||
path = _salt_path()
|
||||
# Read before writing. Hooks are separate processes firing on every tool
|
||||
# call, so all but the first find the salt already published; going straight
|
||||
# to create-fsync-link-unlink meant every one of them paid an fsync to
|
||||
# discover that, on a path whose whole promise is appending a line and
|
||||
# returning.
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
temporary = path.with_name(f"{path.name}.{os.getpid()}.tmp")
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
stream.write(uuid.uuid4().hex)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
try:
|
||||
# Atomic claim: fails if another process already published one.
|
||||
# os.link rather than replace, which would clobber theirs.
|
||||
os.link(temporary, path)
|
||||
except FileExistsError:
|
||||
pass
|
||||
except OSError:
|
||||
# No hardlinks here (some network mounts, some container volumes).
|
||||
# Claim the name directly instead. That reopens the empty-file
|
||||
# window, but the window is now benign: a reader that lands in it
|
||||
# gets "" and omits the hash for that process rather than caching a
|
||||
# guessable one. Losing the hashes on every run of an entire
|
||||
# filesystem is the worse failure.
|
||||
try:
|
||||
fallback = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fallback, "w", encoding="utf-8") as stream:
|
||||
stream.write(temporary.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_salt_cache = ""
|
||||
return _salt_cache
|
||||
|
||||
|
||||
def _scoped_digest(value: str, length: int = 16) -> str:
|
||||
"""Salted digest for values drawn from a guessable space.
|
||||
|
||||
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
||||
no remote — which normally contains the account username. Sixteen unsalted
|
||||
hex characters over that input space is enumerable, so this is not a
|
||||
privacy control without the salt. Salting per install keeps every
|
||||
within-account join the analytics actually use and gives up only
|
||||
cross-machine joins on the same repository, which nothing computes.
|
||||
|
||||
Returns "" when there is no salt, so record() omits the property. An
|
||||
unsalted digest over this input space is close to plaintext, and emitting one
|
||||
under a name that implies it is hashed is worse than sending nothing.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
salt = _install_salt()
|
||||
if not salt:
|
||||
return ""
|
||||
return hashlib.sha256(f"{salt}:{value}".encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _safe_value(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return memory_core.redact(value)
|
||||
@@ -145,9 +302,176 @@ def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
||||
return created
|
||||
|
||||
|
||||
def _rotate_anonymous_id(identity: dict[str, str]) -> str:
|
||||
"""Mint a fresh anonymous id because the account context is gone.
|
||||
|
||||
The previous id may already have been merged into a person profile by an
|
||||
$identify, and that merge is permanent. Reusing it after a logout or a key
|
||||
change attributes everything that follows to the account that just went
|
||||
away, which is the same misattribution the key fingerprint exists to stop,
|
||||
only arriving through the anonymous path instead.
|
||||
|
||||
`aliased` is cleared with it: the new id has never been merged, so it is
|
||||
eligible to be aliased into whatever account comes next.
|
||||
"""
|
||||
created = f"code-anon-{uuid.uuid4().hex}"
|
||||
identity["anonymous_id"] = created
|
||||
identity.pop("aliased", None)
|
||||
_write_identity(identity)
|
||||
return created
|
||||
|
||||
|
||||
def _install_state_path() -> Path:
|
||||
return memory_core.data_dir() / "install-state.json"
|
||||
|
||||
|
||||
def is_first_run() -> bool:
|
||||
"""Whether this machine has never recorded a plugin event before."""
|
||||
return not _identity_path().exists()
|
||||
"""Whether install has never been recorded on this machine.
|
||||
|
||||
Deliberately NOT the identity file. That file is only written by a
|
||||
successful flush, so an offline or firewalled user recorded code.install on
|
||||
every single session, forever — and every 0.2.x user recorded one on their
|
||||
first 0.3.x session because 0.2.x never wrote it at all.
|
||||
"""
|
||||
return not _install_state_path().exists()
|
||||
|
||||
|
||||
def data_dir_was_empty() -> bool:
|
||||
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
||||
|
||||
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
||||
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
||||
asking at claim time always saw content and every fresh install reported an
|
||||
upgrade. The caller snapshots this at the top of the run instead.
|
||||
"""
|
||||
return not _data_dir_has_content()
|
||||
|
||||
|
||||
def claim_install(was_empty: bool | None = None) -> str | None:
|
||||
"""Claim the one install/upgrade record for this machine, atomically.
|
||||
|
||||
Returns the event to record ("install" or "upgrade"), or None if another
|
||||
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
||||
cannot both win.
|
||||
|
||||
`was_empty` must come from data_dir_was_empty() called before this process
|
||||
wrote anything. Omitting it falls back to checking now, which is only
|
||||
correct for a caller that has touched nothing.
|
||||
"""
|
||||
if not is_enabled():
|
||||
# Never consume the one-shot claim while the user is opted out, or they
|
||||
# would silently lose their install event if they later opt in.
|
||||
return None
|
||||
|
||||
path = _install_state_path()
|
||||
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
json.dump(
|
||||
{
|
||||
"plugin_version": memory_core.PLUGIN_VERSION,
|
||||
"installed_at": memory_core.utc_now(),
|
||||
"upgraded": upgrading,
|
||||
},
|
||||
stream,
|
||||
)
|
||||
# Durable before this returns. The O_EXCL open is what makes the
|
||||
# claim exclusive, so it cannot be replaced by a temp-and-rename
|
||||
# without losing that, which leaves the content as the thing to make
|
||||
# safe. A kill between the open and this fsync used to leave a marker
|
||||
# that exists but parses to nothing: is_first_run reads it as claimed
|
||||
# and claim_version_change cannot read a version out of it.
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
except OSError:
|
||||
pass
|
||||
return "upgrade" if upgrading else "install"
|
||||
|
||||
|
||||
def _data_dir_has_content() -> bool:
|
||||
"""Whether anything predates this session in the plugin data directory."""
|
||||
try:
|
||||
for entry in memory_core.data_dir().iterdir():
|
||||
if entry.name != "install-state.json":
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _repair_install_state(path: Path) -> None:
|
||||
"""Rewrite an unparseable marker so version tracking can resume."""
|
||||
try:
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
temporary.write_text(
|
||||
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def claim_version_change() -> str | None:
|
||||
"""Return the previously recorded version if it differs, updating the marker.
|
||||
|
||||
Only meaningful once the marker exists — the first transition into 0.3.x has
|
||||
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
||||
the marker means the next session sees no change and records nothing.
|
||||
"""
|
||||
path = _install_state_path()
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
return None
|
||||
except json.JSONDecodeError:
|
||||
# A crash between O_EXCL and the write leaves an empty marker. Left
|
||||
# alone it disables every future upgrade event on this machine, because
|
||||
# claim_install sees the file and this function cannot parse it.
|
||||
state = None
|
||||
if not isinstance(state, dict):
|
||||
_repair_install_state(path)
|
||||
return None
|
||||
previous = str(state.get("plugin_version") or "")
|
||||
if not previous or previous == memory_core.PLUGIN_VERSION:
|
||||
return None
|
||||
# Claim the transition with an exclusive sentinel before rewriting the
|
||||
# marker. A plain read-modify-write let every concurrently starting session
|
||||
# observe the old version and each record its own upgrade — and the first
|
||||
# session after a version bump is exactly when several agent windows restart
|
||||
# together.
|
||||
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
||||
try:
|
||||
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
||||
state["upgraded_at"] = memory_core.utc_now()
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
try:
|
||||
temporary.write_text(json.dumps(state), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
# Release the claim. The marker still records the old version, so
|
||||
# without this the sentinel makes claim_version_change return early on
|
||||
# every later run and this version's upgrade is never recorded again.
|
||||
for leftover in (sentinel, temporary):
|
||||
try:
|
||||
leftover.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
return previous
|
||||
|
||||
|
||||
def record(
|
||||
@@ -168,19 +492,32 @@ def record(
|
||||
except OSError:
|
||||
pass
|
||||
properties = _safe_value(properties)
|
||||
# Stamped in the RECORDING process, beside harness. `source` used to be
|
||||
# read in the sending process from a module global, so whichever process
|
||||
# drained the spool named every event in it. flush() spreads per-event
|
||||
# properties last, so this now wins over any sender's default.
|
||||
properties.update(
|
||||
harness=_harness,
|
||||
source=_source_tag,
|
||||
plugin_version=memory_core.PLUGIN_VERSION,
|
||||
os=sys.platform,
|
||||
python_version=platform.python_version(),
|
||||
)
|
||||
# Assigned only when the digest is real. _scoped_digest returns "" when
|
||||
# the salt could not be persisted, and an empty property is worse than an
|
||||
# absent one: it survives the None filter below and reads as a value.
|
||||
if repo is not None:
|
||||
properties["repo_hash"] = _digest(getattr(repo, "identity", ""))
|
||||
repo_hash = _scoped_digest(getattr(repo, "identity", ""))
|
||||
if repo_hash:
|
||||
properties["repo_hash"] = repo_hash
|
||||
if session_id:
|
||||
properties["session_hash"] = _digest(session_id)
|
||||
session_hash = _scoped_digest(session_id)
|
||||
if session_hash:
|
||||
properties["session_hash"] = session_hash
|
||||
line = json.dumps(
|
||||
{
|
||||
"event": f"{EVENT_PREFIX}.{event}",
|
||||
"uuid": str(uuid.uuid4()),
|
||||
"timestamp": memory_core.utc_now(),
|
||||
"properties": {
|
||||
key: value for key, value in properties.items() if value is not None
|
||||
@@ -239,38 +576,201 @@ def spawn_flush() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _claim_name(attempt: int = 0) -> str:
|
||||
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
||||
only ever discards a batch that was actually retried and failed."""
|
||||
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
||||
|
||||
|
||||
def _claim_attempt(claim: Path) -> int:
|
||||
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape.
|
||||
|
||||
Anchored on field position, not on a leading "a": the legacy shape is
|
||||
``telemetry-<pid>-<hex>.sending`` and a hex id such as ``a1234567`` would
|
||||
otherwise parse as attempt 1234567 and be discarded unsent on the first
|
||||
flush after an upgrade.
|
||||
"""
|
||||
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
||||
parts = stem.split("-")
|
||||
if len(parts) != 4:
|
||||
return 0
|
||||
tail = parts[3]
|
||||
if tail.startswith("a") and tail[1:].isdigit():
|
||||
return int(tail[1:])
|
||||
return 0
|
||||
|
||||
|
||||
def _touch(path: Path) -> None:
|
||||
"""Refresh mtime so a claim's age measures time since it was claimed.
|
||||
|
||||
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
||||
after a quiet minute inherited the spool's last-write time and looked
|
||||
abandoned the instant it was made. A second sender would then take it over
|
||||
while the first was still posting, and both would deliver the batch.
|
||||
"""
|
||||
try:
|
||||
os.utime(path, None)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _claim_spool() -> Path | None:
|
||||
"""Rename the spool aside so exactly one sender owns each batch."""
|
||||
directory = memory_core.data_dir()
|
||||
claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending"
|
||||
claim = directory / _claim_name()
|
||||
spool = _spool_path()
|
||||
try:
|
||||
spool.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
pass
|
||||
return _claim_parked(directory)
|
||||
|
||||
|
||||
def _sweep_debris(directory: Path) -> None:
|
||||
"""Remove files nothing else will ever pick up again.
|
||||
|
||||
*.partial is a temp file orphaned by a crash between write and rename.
|
||||
*.corrupt is a batch quarantined for undecodable content. No glob in this
|
||||
module matches either, so without this they accumulate on disk for the life
|
||||
of the install.
|
||||
|
||||
Quarantined batches are kept far longer than debris: they are the only
|
||||
evidence left of events that could not be delivered, and someone diagnosing
|
||||
a report of missing telemetry has to be able to find one.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending")):
|
||||
for debris in directory.glob("telemetry-*.partial"):
|
||||
try:
|
||||
if now - debris.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
debris.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
for quarantined in directory.glob("telemetry-*.corrupt"):
|
||||
try:
|
||||
if now - quarantined.stat().st_mtime > CLAIM_EXPIRY_SECONDS:
|
||||
quarantined.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
# The same reasoning covers *.tmp. _write_identity and _install_salt both
|
||||
# create one and unlink it in a finally, which a SIGKILL skips, and no glob
|
||||
# in this module matches the leftovers either.
|
||||
for temporary in directory.glob("telemetry-*.tmp"):
|
||||
try:
|
||||
if now - temporary.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
|
||||
def _claim_parked(directory: Path) -> Path | None:
|
||||
"""Take the oldest abandoned claim, if any lease has actually expired.
|
||||
|
||||
Kept separate from the live spool so flush() can drain both in one run.
|
||||
Previously parked batches were only reachable when no spool existed at all,
|
||||
and because sessions keep recording there usually was one — so a batch
|
||||
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
||||
even though its own presence is what started the sender.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
||||
try:
|
||||
age = now - orphan.stat().st_mtime
|
||||
except OSError:
|
||||
continue
|
||||
if age > CLAIM_EXPIRY_SECONDS:
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
# Someone else holds a live lease on it. This check has to come
|
||||
# first. Claiming a file bumps its attempt count and refreshes its
|
||||
# mtime, so a sender that has just taken the final attempt looks
|
||||
# exhausted to everyone else while it is actively draining. Judging
|
||||
# exhaustion before liveness let a second sender unlink a batch out
|
||||
# from under its owner, losing every event in it.
|
||||
continue
|
||||
# Attempts, not age. Every re-claim touches the mtime and every release
|
||||
# backdates it by a fixed amount, so age is pinned near the stale
|
||||
# threshold and never reaches the expiry. Age stays only as a backstop
|
||||
# for files that never carried an attempt marker.
|
||||
if _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS or age > CLAIM_EXPIRY_SECONDS:
|
||||
try:
|
||||
orphan.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
continue
|
||||
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
||||
try:
|
||||
orphan.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _safe_mtime(path: Path) -> float:
|
||||
try:
|
||||
return path.stat().st_mtime
|
||||
except OSError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
||||
"""Persist the unsent remainder, atomically, and refresh the lease.
|
||||
|
||||
Called after every successful batch. Two jobs: a retry resumes where the
|
||||
send stopped instead of re-posting from the top, and the rewrite doubles as
|
||||
the lease heartbeat, so a slow sender does not have its claim stolen
|
||||
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
||||
"""
|
||||
if not remaining:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return True
|
||||
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
||||
try:
|
||||
payload = "".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining)
|
||||
# fsync before the rename: without it the rename can land while the
|
||||
# bytes have not, and the claim comes back empty or truncated after a
|
||||
# crash. _drain then reads zero events and unlinks it.
|
||||
with open(temporary, "w", encoding="utf-8") as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
temporary.replace(claim)
|
||||
_touch(claim)
|
||||
return True
|
||||
except OSError:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
||||
"""Persist the remainder and drop the lease, because this sender has given up.
|
||||
|
||||
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
||||
make an abandoned batch look actively owned for a further
|
||||
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
||||
threshold lets the next flush pick it up immediately, while the attempt
|
||||
count in the filename still bounds how many times that can happen.
|
||||
"""
|
||||
if not _rewrite_claim(claim, remaining):
|
||||
return
|
||||
try:
|
||||
# Backdate past the stale threshold so the next flush can pick it up,
|
||||
# minus a cooldown that grows with the attempts already spent. Clamped so
|
||||
# the mtime never lands in the future, which would read as a live lease.
|
||||
cooldown = min(_claim_attempt(claim) * RETRY_COOLDOWN_SECONDS, CLAIM_STALE_SECONDS)
|
||||
released = time.time() - CLAIM_STALE_SECONDS - 1 + cooldown
|
||||
os.utime(claim, (released, released))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_email(key: str) -> str:
|
||||
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
||||
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
||||
@@ -300,34 +800,130 @@ def _post(payload: dict[str, Any], url: str) -> bool:
|
||||
|
||||
|
||||
def resolve_distinct_id() -> tuple[str, str]:
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any."""
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
||||
|
||||
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
||||
anonymous id: aliasing one account email to another merges two real person
|
||||
profiles and cannot be undone, so a key that now belongs to a different
|
||||
account re-resolves with no alias.
|
||||
"""
|
||||
identity = _read_identity()
|
||||
email = identity.get("email", "")
|
||||
if email:
|
||||
return email, ""
|
||||
key = memory_core.api_key()
|
||||
fingerprint = _digest(key) if key else ""
|
||||
email = identity.get("email", "")
|
||||
|
||||
if email and fingerprint:
|
||||
recorded = identity.get("key_fingerprint", "")
|
||||
if recorded == fingerprint:
|
||||
return email, ""
|
||||
if not recorded:
|
||||
# Rows written before fingerprints existed. Verify rather than
|
||||
# adopt: a key changed before the upgrade would otherwise bind the
|
||||
# new key to the previous account's email, permanently, and the
|
||||
# fingerprint would then agree with itself forever after.
|
||||
verified = _resolve_email(key)
|
||||
if not verified:
|
||||
# Offline, firewalled, or the API is down. Keep the previous
|
||||
# behaviour and retry on the next flush rather than dropping a
|
||||
# real account attribution. Safe because the same network that
|
||||
# failed /v1/ping/ is about to fail the PostHog POST, so nothing
|
||||
# is delivered under the unverified identity in the meantime.
|
||||
return email, ""
|
||||
identity["email"] = verified
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return verified, ""
|
||||
|
||||
if not key:
|
||||
# No key to verify the account with; do not keep attributing to it.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
email = _resolve_email(key)
|
||||
if not email:
|
||||
|
||||
resolved = _resolve_email(key)
|
||||
if not resolved:
|
||||
# The key changed and will not resolve (revoked, offline, API down).
|
||||
# Reaching here with an email means the recorded fingerprint disagreed,
|
||||
# so the key really did change. Drop the account and rotate: the stored
|
||||
# anonymous id may already be merged into that account's person, and
|
||||
# reusing it would keep the events on the profile we are trying to
|
||||
# leave.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
previous = identity.get("anonymous_id", "")
|
||||
identity["email"] = email
|
||||
|
||||
# Alias only when going anonymous -> email for the first time. Once an anon
|
||||
# id has been merged into an account it must never be offered again: an
|
||||
# alias naming an already-identified id is what could link two real people.
|
||||
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
||||
if previous:
|
||||
identity["aliased"] = True
|
||||
identity["email"] = resolved
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return email, previous
|
||||
return resolved, previous
|
||||
|
||||
|
||||
def flush() -> int:
|
||||
"""Drain claimed spools to PostHog and return the number of events sent."""
|
||||
"""Drain the live spool, then any parked claims, and return events sent."""
|
||||
if not is_enabled():
|
||||
return 0
|
||||
claim = _claim_spool()
|
||||
sent, delivered = _drain(_claim_spool())
|
||||
if not delivered:
|
||||
# The network is failing. Retrying other batches now would only burn
|
||||
# their attempt budget against the same broken connection.
|
||||
return sent
|
||||
|
||||
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
||||
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
||||
directory = memory_core.data_dir()
|
||||
_sweep_debris(directory)
|
||||
for _ in range(MAX_PARKED_PER_RUN):
|
||||
parked = _claim_parked(directory)
|
||||
if parked is None:
|
||||
break
|
||||
count, delivered = _drain(parked)
|
||||
sent += count
|
||||
if not delivered:
|
||||
break
|
||||
return sent
|
||||
|
||||
|
||||
def _drain(claim: Path | None) -> tuple[int, bool]:
|
||||
"""Post one claimed batch file, recording progress after every batch.
|
||||
|
||||
Returns (events sent, whether everything was delivered).
|
||||
"""
|
||||
if claim is None:
|
||||
return 0
|
||||
return 0, True
|
||||
try:
|
||||
lines = claim.read_text(encoding="utf-8").splitlines()
|
||||
except ValueError:
|
||||
# UnicodeDecodeError from a torn write: the content is unrecoverable, so
|
||||
# quarantine rather than retry. flush() runs from a bare `finally:` in
|
||||
# flush_worker, so raising here also skips the handoff cleanup, and an
|
||||
# undecodable file would otherwise be re-read on every flush forever.
|
||||
# Reported as delivered because there is nothing left to deliver and the
|
||||
# rest of the run should continue.
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt"))
|
||||
except OSError:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0, True
|
||||
except OSError:
|
||||
return 0
|
||||
# Could not read it, which is not the same as having nothing to send.
|
||||
# The file is left exactly where it is: a vanished or briefly unreadable
|
||||
# claim is retryable, and quarantining it here would discard events over
|
||||
# a transient filesystem error. Reported as undelivered so the run stops
|
||||
# instead of counting a batch nothing was posted from as delivered.
|
||||
return 0, False
|
||||
events = []
|
||||
for line in lines:
|
||||
try:
|
||||
@@ -337,11 +933,18 @@ def flush() -> int:
|
||||
if isinstance(value, dict) and value.get("event"):
|
||||
events.append(value)
|
||||
if not events:
|
||||
# Only delete when the file really is empty. A non-empty file that
|
||||
# parses to nothing is a torn write, and its contents are the unsent
|
||||
# remainder — deleting it is the data loss this PR exists to prevent.
|
||||
try:
|
||||
claim.unlink()
|
||||
empty = claim.stat().st_size == 0
|
||||
except OSError:
|
||||
empty = True
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt")) if not empty else claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
return 0, True
|
||||
|
||||
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
||||
if aliased_anonymous_id:
|
||||
@@ -360,12 +963,17 @@ def flush() -> int:
|
||||
|
||||
sent = 0
|
||||
for start in range(0, len(events), BATCH_SIZE):
|
||||
chunk = events[start : start + BATCH_SIZE]
|
||||
batch = [
|
||||
{
|
||||
"event": event["event"],
|
||||
"distinct_id": distinct_id,
|
||||
# Carried through from record() so a resend can be collapsed.
|
||||
"uuid": event.get("uuid"),
|
||||
"timestamp": event.get("timestamp"),
|
||||
"properties": {
|
||||
# Fallback only: events recorded by a build before source
|
||||
# moved into record() have none of their own.
|
||||
"source": _source_tag,
|
||||
"language": "python",
|
||||
"$process_person_profile": False,
|
||||
@@ -373,16 +981,24 @@ def flush() -> int:
|
||||
**(event.get("properties") or {}),
|
||||
},
|
||||
}
|
||||
for event in events[start : start + BATCH_SIZE]
|
||||
for event in chunk
|
||||
]
|
||||
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
||||
return sent
|
||||
sent += len(batch)
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return sent
|
||||
# Keep only what has not been delivered, and release the lease.
|
||||
# Previously the whole file was kept and the retry re-posted every
|
||||
# batch, including the ones that had already arrived.
|
||||
_release_claim(claim, events[start:])
|
||||
return sent, False
|
||||
sent += len(chunk)
|
||||
# Record progress and refresh the lease after each successful batch, so
|
||||
# a crash repeats at most one batch instead of the entire file. If the
|
||||
# rewrite fails the claim still holds delivered events, so stop rather
|
||||
# than carry on as though progress were recorded — continuing is how the
|
||||
# duplicate delivery this PR fixes would come back.
|
||||
if not _rewrite_claim(claim, events[start + len(chunk) :]):
|
||||
_release_claim(claim, events[start + len(chunk) :])
|
||||
return sent, False
|
||||
return sent, True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
||||
@@ -7,8 +7,8 @@ disable-model-invocation: true
|
||||
# Pause memory capture
|
||||
|
||||
To pause (hooks stop capturing and sending session content; a minimal
|
||||
anonymous telemetry ping still fires at session start unless
|
||||
`MEM0_TELEMETRY=false`):
|
||||
telemetry ping still fires at session start, under your Mem0 account email,
|
||||
unless `MEM0_TELEMETRY=false`):
|
||||
|
||||
```bash
|
||||
python3 "{{PLUGIN_ROOT}}/core/memory_cli.py" --harness "{{HARNESS_ID}}" {{PLUGIN_DATA_ARG}} pause
|
||||
|
||||
@@ -70,6 +70,38 @@ def test_portable_bundle_is_conformant_and_self_contained(tmp_path: Path) -> Non
|
||||
assert not any(path.is_symlink() for path in root.rglob("*"))
|
||||
|
||||
|
||||
def _harness_identity(root: Path) -> dict[str, str]:
|
||||
"""Read the generated core/_harness_id.py without importing it."""
|
||||
values: dict[str, str] = {}
|
||||
for line in (root / "core" / "_harness_id.py").read_text(encoding="utf-8").splitlines():
|
||||
if "=" in line and not line.lstrip().startswith("#"):
|
||||
name, _, raw = line.partition("=")
|
||||
values[name.strip()] = raw.strip().strip('"')
|
||||
return values
|
||||
|
||||
|
||||
def test_the_portable_bundle_declares_no_host_application(tmp_path: Path) -> None:
|
||||
"""It runs in whatever editor a user drops it into, so it cannot know the host.
|
||||
|
||||
X-Application is allowlisted server-side. A guessed value is silently dropped
|
||||
there, which is the worst outcome: the wire says we know the host and the
|
||||
stored event says we do not.
|
||||
"""
|
||||
identity = _harness_identity(build("mem0-agent-plugin", "portable", tmp_path / "portable"))
|
||||
|
||||
assert identity["PLATFORM_APPLICATION"] == ""
|
||||
# The PostHog-side label is still useful for grouping and stays populated.
|
||||
assert identity["HARNESS_ID"] == "coding-agent"
|
||||
assert identity["PLATFORM_SOURCE"] == "MEM0_PLUGIN"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("host", ["claude-code", "cursor", "codex", "kimi", "antigravity"])
|
||||
def test_a_native_bundle_names_the_host_it_was_built_for(host: str, tmp_path: Path) -> None:
|
||||
identity = _harness_identity(build(host, "native", tmp_path / host))
|
||||
|
||||
assert identity["PLATFORM_APPLICATION"] == host
|
||||
|
||||
|
||||
@pytest.mark.parametrize("host", ["claude-code", "cursor", "codex", "kimi", "antigravity"])
|
||||
def test_native_bundle_is_self_contained(host: str, tmp_path: Path) -> None:
|
||||
root = build(host, "native", tmp_path / host)
|
||||
|
||||
@@ -0,0 +1,446 @@
|
||||
"""Delivery semantics of the telemetry spool: no duplicates, no starvation.
|
||||
|
||||
These run against a built host's core in-process (not a subprocess) because they
|
||||
need to inject failures into ``_post``. The identity tests next door cover the
|
||||
uninitialised-process case that needs a real interpreter.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
CORE_ROOT = Path(__file__).resolve().parents[1]
|
||||
REPOSITORY_ROOT = CORE_ROOT.parents[1]
|
||||
HOST_CORE = REPOSITORY_ROOT / "integrations" / "claude-code-plugin" / "core"
|
||||
|
||||
pytestmark = pytest.mark.skipif(not HOST_CORE.exists(), reason="claude-code-plugin is not built")
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def telemetry(tmp_path, monkeypatch):
|
||||
# CI runs this directory and claude-code-plugin/tests in ONE pytest process,
|
||||
# and that suite's conftest sets MEM0_TELEMETRY=false at import, process-wide.
|
||||
# Without this the whole file silently no-ops: record() returns early and
|
||||
# every assertion sees an empty spool. Do not rely on ambient env.
|
||||
monkeypatch.setenv("MEM0_TELEMETRY", "true")
|
||||
monkeypatch.setenv("MEM0_CODE_DATA_DIR", str(tmp_path / "data"))
|
||||
monkeypatch.syspath_prepend(str(HOST_CORE))
|
||||
|
||||
# Save and RESTORE rather than delete. claude-code-plugin/tests/conftest.py
|
||||
# imports memory_core once at collection and calls configure_harness() on it;
|
||||
# dropping the module left a later re-import with default harness config, so
|
||||
# tests in that suite failed depending on collection order.
|
||||
names = ("telemetry", "memory_core", "_harness_id")
|
||||
saved = {name: sys.modules.get(name) for name in names}
|
||||
for name in names:
|
||||
sys.modules.pop(name, None)
|
||||
|
||||
module = importlib.import_module("telemetry")
|
||||
monkeypatch.setattr(module, "resolve_distinct_id", lambda: ("tester@example.com", ""))
|
||||
try:
|
||||
yield module
|
||||
finally:
|
||||
for name in names:
|
||||
sys.modules.pop(name, None)
|
||||
if saved[name] is not None:
|
||||
sys.modules[name] = saved[name]
|
||||
|
||||
|
||||
def _delivered(payloads):
|
||||
return [event for payload in payloads if "batch" in payload for event in payload["batch"]]
|
||||
|
||||
|
||||
def test_a_partial_failure_does_not_redeliver_what_already_arrived(telemetry):
|
||||
"""Defect 2a: flush kept the whole claim on failure and retried from the top.
|
||||
|
||||
150 events across two batches, the second failing, previously delivered 250.
|
||||
"""
|
||||
for index in range(150):
|
||||
telemetry.record("search", index=index)
|
||||
|
||||
sent: list[dict] = []
|
||||
calls = {"n": 0}
|
||||
|
||||
def flaky(payload, url):
|
||||
calls["n"] += 1
|
||||
if calls["n"] == 2: # second batch fails
|
||||
return False
|
||||
sent.append(payload)
|
||||
return True
|
||||
|
||||
telemetry._post = flaky
|
||||
telemetry.flush()
|
||||
|
||||
telemetry._post = lambda payload, url: sent.append(payload) or True
|
||||
telemetry.flush()
|
||||
|
||||
events = _delivered(sent)
|
||||
assert len(events) == 150
|
||||
assert len({event["uuid"] for event in events}) == 150
|
||||
|
||||
|
||||
def test_a_fresh_claim_is_not_immediately_stealable(telemetry):
|
||||
"""Defect 2b: rename preserves mtime, so a claim inherited the spool's age.
|
||||
|
||||
With the last write older than the stale threshold, a claim made now looked
|
||||
abandoned the instant it existed and a second sender took it over.
|
||||
"""
|
||||
telemetry.record("search")
|
||||
spool = telemetry._spool_path()
|
||||
old = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(spool, (old, old))
|
||||
|
||||
first = telemetry._claim_spool()
|
||||
assert first is not None
|
||||
|
||||
# A second sender starting right now must find nothing to take.
|
||||
assert telemetry._claim_parked(first.parent) is None
|
||||
|
||||
|
||||
def test_a_live_final_attempt_is_not_deleted_by_another_sender(telemetry):
|
||||
"""Review finding: exhaustion was judged before liveness, so owners lost batches.
|
||||
|
||||
Claiming a parked file bumps its attempt count and refreshes its mtime. Once
|
||||
the count reaches the budget, the owner draining it looked exhausted to every
|
||||
other sender, which unlinked the file out from under it. Everything in that
|
||||
batch was gone, which is precisely the loss this PR exists to stop.
|
||||
"""
|
||||
telemetry.record("search", reason="owned-by-the-first-sender")
|
||||
spool = telemetry._spool_path()
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(spool, (stale, stale))
|
||||
|
||||
claim = telemetry._claim_spool()
|
||||
assert claim is not None
|
||||
|
||||
# Walk it to the final attempt, ageing it each round so it can be re-claimed.
|
||||
# _claim_spool hands back a0 and _release_claim keeps the name, so it takes
|
||||
# one full round per attempt to reach the budget.
|
||||
for _ in range(telemetry.MAX_CLAIM_ATTEMPTS):
|
||||
# Carry the marker through each rewrite so the final assertion proves the
|
||||
# events survived, not merely that some file with the right name did.
|
||||
telemetry._release_claim(claim, [{"event": "code.search", "uuid": "owned-by-the-first-sender"}])
|
||||
parked = sorted(claim.parent.glob("telemetry-*.sending"))
|
||||
assert parked, "the batch was dropped while still inside its budget"
|
||||
os.utime(parked[0], (stale, stale))
|
||||
claim = telemetry._claim_parked(claim.parent)
|
||||
assert claim is not None
|
||||
|
||||
assert telemetry._claim_attempt(claim) >= telemetry.MAX_CLAIM_ATTEMPTS
|
||||
assert claim.exists()
|
||||
|
||||
# The owner is draining it right now: fresh mtime, live lease.
|
||||
second_sender = telemetry._claim_parked(claim.parent)
|
||||
|
||||
assert second_sender is None, "a second sender took a batch under a live lease"
|
||||
assert claim.exists(), "a second sender deleted a batch its owner was draining"
|
||||
assert "owned-by-the-first-sender" in claim.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_an_exhausted_batch_is_still_discarded_once_its_lease_lapses(telemetry):
|
||||
"""The liveness check must defer the cleanup, not cancel it.
|
||||
|
||||
Guards the obvious over-correction: skipping live claims is only safe if an
|
||||
abandoned one at the same attempt count is still reaped on a later run.
|
||||
"""
|
||||
telemetry.record("search")
|
||||
spool = telemetry._spool_path()
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(spool, (stale, stale))
|
||||
|
||||
claim = telemetry._claim_spool()
|
||||
assert claim is not None
|
||||
exhausted = claim.parent / telemetry._claim_name(telemetry.MAX_CLAIM_ATTEMPTS)
|
||||
claim.replace(exhausted)
|
||||
os.utime(exhausted, (stale, stale))
|
||||
|
||||
assert telemetry._claim_parked(exhausted.parent) is None
|
||||
assert not exhausted.exists(), "an abandoned exhausted batch was left behind forever"
|
||||
|
||||
|
||||
def test_a_parked_batch_is_drained_behind_the_live_spool(telemetry):
|
||||
"""Defect 6: parked claims were only reachable when no spool existed.
|
||||
|
||||
Because sessions keep recording there usually was one, so a batch parked by
|
||||
a failed send waited until the 7-day expiry deleted it unsent — even though
|
||||
its own presence is what starts the sender.
|
||||
"""
|
||||
telemetry.record("parked")
|
||||
telemetry._post = lambda payload, url: False
|
||||
telemetry.flush()
|
||||
|
||||
parked = list(telemetry.memory_core.data_dir().glob("telemetry-*.sending"))
|
||||
assert len(parked) == 1
|
||||
old = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(parked[0], (old, old))
|
||||
|
||||
telemetry.record("fresh")
|
||||
sent: list[dict] = []
|
||||
telemetry._post = lambda payload, url: sent.append(payload) or True
|
||||
telemetry.flush()
|
||||
|
||||
names = {event["event"] for event in _delivered(sent)}
|
||||
assert names == {"code.parked", "code.fresh"}
|
||||
|
||||
|
||||
def test_a_batch_is_retried_until_the_budget_is_spent_not_discarded(telemetry):
|
||||
"""Expiry discards what failed repeatedly, not what merely sat for a while.
|
||||
|
||||
The budget is the attempt count, because age cannot be one: every re-claim
|
||||
touches the mtime and every release backdates it, so age never accumulates.
|
||||
"""
|
||||
telemetry.record("parked")
|
||||
telemetry._post = lambda payload, url: False
|
||||
telemetry.flush()
|
||||
|
||||
parked = list(telemetry.memory_core.data_dir().glob("telemetry-*.sending"))
|
||||
assert len(parked) == 1
|
||||
assert telemetry._claim_attempt(parked[0]) < telemetry.MAX_CLAIM_ATTEMPTS
|
||||
|
||||
sent: list[dict] = []
|
||||
telemetry._post = lambda payload, url: sent.append(payload) or True
|
||||
telemetry.flush()
|
||||
|
||||
assert [event["event"] for event in _delivered(sent)] == ["code.parked"]
|
||||
|
||||
|
||||
def test_progress_is_recorded_after_every_batch(telemetry):
|
||||
"""A crash repeats at most one batch, not the whole file."""
|
||||
for index in range(250):
|
||||
telemetry.record("search", index=index)
|
||||
|
||||
calls = {"n": 0}
|
||||
|
||||
def die_after_two(payload, url):
|
||||
calls["n"] += 1
|
||||
if calls["n"] > 2:
|
||||
return False
|
||||
return True
|
||||
|
||||
telemetry._post = die_after_two
|
||||
telemetry.flush()
|
||||
|
||||
parked = list(telemetry.memory_core.data_dir().glob("telemetry-*.sending"))
|
||||
assert len(parked) == 1
|
||||
remaining = parked[0].read_text(encoding="utf-8").strip().splitlines()
|
||||
# Two batches of 100 landed; only the last 50 should still be pending.
|
||||
assert len(remaining) == 50
|
||||
assert json.loads(remaining[0])["properties"]["index"] == 200
|
||||
|
||||
|
||||
def test_the_heartbeat_actually_refreshes_the_lease(telemetry):
|
||||
"""The claim rewrite doubles as the lease heartbeat.
|
||||
|
||||
Previously asserted `SEND_TIMEOUT * 4 < CLAIM_STALE_SECONDS`, which compares
|
||||
two constants and executes none of the code under test. Drive the real
|
||||
rewrite and watch the mtime move instead.
|
||||
"""
|
||||
for index in range(150):
|
||||
telemetry.record("search", index=index)
|
||||
claim = telemetry._claim_spool()
|
||||
assert claim is not None
|
||||
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(claim, (stale, stale))
|
||||
assert time.time() - claim.stat().st_mtime > telemetry.CLAIM_STALE_SECONDS
|
||||
|
||||
telemetry._rewrite_claim(claim, [{"event": "code.x", "properties": {}}])
|
||||
assert time.time() - claim.stat().st_mtime < telemetry.CLAIM_STALE_SECONDS
|
||||
|
||||
|
||||
def test_an_undeliverable_batch_is_eventually_given_up_on(telemetry):
|
||||
"""Expiry has to be reachable from a state the state machine can produce.
|
||||
|
||||
It was not: every re-claim touched the mtime and every release backdated it
|
||||
by a fixed amount, so age hovered near the stale threshold and the 7-day
|
||||
expiry never fired. An undeliverable batch lived on disk forever, and
|
||||
spawn_flush saw it and started a sender on every hook.
|
||||
"""
|
||||
telemetry.record("doomed")
|
||||
telemetry._post = lambda payload, url: False
|
||||
|
||||
directory = telemetry.memory_core.data_dir()
|
||||
for _ in range(telemetry.MAX_CLAIM_ATTEMPTS + 3):
|
||||
telemetry.flush()
|
||||
# Attempts now carry a cooldown, so a released claim is not instantly
|
||||
# reclaimable. Age it to stand in for the wall time a real retry waits;
|
||||
# without this the loop spins inside one cooldown and proves nothing.
|
||||
for parked in directory.glob("telemetry-*.sending"):
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(parked, (stale, stale))
|
||||
|
||||
leftover = list(directory.glob("telemetry-*.sending"))
|
||||
assert leftover == [], f"batch never given up on: {[p.name for p in leftover]}"
|
||||
|
||||
|
||||
def test_a_batch_that_cannot_be_read_is_not_counted_as_delivered(telemetry):
|
||||
"""Review finding: a read failure reported 'everything delivered'.
|
||||
|
||||
Nothing was posted, so calling it delivered lets flush() carry on to other
|
||||
claims as though this batch had arrived, and hides the failure from the one
|
||||
signal that says the run went badly. It also must not quarantine: a briefly
|
||||
unreadable file is retryable, and moving it to .corrupt discards the events
|
||||
over a transient filesystem error, because nothing ever re-globs .corrupt.
|
||||
"""
|
||||
telemetry.record("search")
|
||||
spool = telemetry._spool_path()
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(spool, (stale, stale))
|
||||
claim = telemetry._claim_spool()
|
||||
assert claim is not None
|
||||
|
||||
original = Path.read_text
|
||||
|
||||
def unreadable(self, *args, **kwargs):
|
||||
if self == claim:
|
||||
raise OSError(5, "I/O error")
|
||||
return original(self, *args, **kwargs)
|
||||
|
||||
Path.read_text = unreadable
|
||||
try:
|
||||
sent, delivered = telemetry._drain(claim)
|
||||
finally:
|
||||
Path.read_text = original
|
||||
|
||||
assert sent == 0
|
||||
assert delivered is False, "an unread batch was reported as delivered"
|
||||
assert claim.exists(), "a transient read error discarded the batch"
|
||||
assert not list(claim.parent.glob("*.corrupt")), "quarantined over a transient error"
|
||||
|
||||
|
||||
def test_undecodable_content_is_still_quarantined_and_the_run_continues(telemetry):
|
||||
"""The other half: genuinely unrecoverable content must not block the run.
|
||||
|
||||
Guards the over-correction. If every read problem returned undelivered, one
|
||||
torn file would stop every later claim on every flush, forever.
|
||||
"""
|
||||
telemetry.record("search")
|
||||
spool = telemetry._spool_path()
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(spool, (stale, stale))
|
||||
claim = telemetry._claim_spool()
|
||||
assert claim is not None
|
||||
claim.write_bytes(b"\xff\xfe torn \x00 write")
|
||||
|
||||
sent, delivered = telemetry._drain(claim)
|
||||
|
||||
assert (sent, delivered) == (0, True)
|
||||
assert not claim.exists()
|
||||
assert list(claim.parent.glob("*.corrupt")), "unrecoverable content was not quarantined"
|
||||
|
||||
|
||||
def test_retries_are_spread_over_real_time_not_burned_at_once(telemetry):
|
||||
"""Review finding: releasing straight to reclaimable spent the budget instantly.
|
||||
|
||||
Two senders hitting one momentary failure could walk a batch from attempt 0
|
||||
to the limit within seconds and discard it, when a retry a minute later would
|
||||
have delivered. Each release now has to age past a cooldown that grows with
|
||||
the attempts already spent.
|
||||
"""
|
||||
telemetry.record("doomed")
|
||||
telemetry._post = lambda payload, url: False
|
||||
|
||||
directory = telemetry.memory_core.data_dir()
|
||||
telemetry.flush()
|
||||
|
||||
parked = list(directory.glob("telemetry-*.sending"))
|
||||
assert parked, "the batch was discarded on its first failure"
|
||||
assert telemetry._claim_attempt(parked[0]) == 0
|
||||
|
||||
# Second sender, immediately: the cooldown has not elapsed, so it must not
|
||||
# be able to spend another attempt.
|
||||
telemetry.flush()
|
||||
still = list(directory.glob("telemetry-*.sending"))
|
||||
assert len(still) == 1
|
||||
assert telemetry._claim_attempt(still[0]) <= 1, "burned attempts without waiting"
|
||||
|
||||
|
||||
def test_a_legacy_claim_filename_is_not_mistaken_for_a_huge_attempt_count(telemetry):
|
||||
"""The old shape is telemetry-<pid>-<hex>.sending, and hex can start with 'a'."""
|
||||
assert telemetry._claim_attempt(Path("telemetry-999-deadbeef.sending")) == 0
|
||||
assert telemetry._claim_attempt(Path("telemetry-999-a1234567.sending")) == 0
|
||||
assert telemetry._claim_attempt(Path("telemetry-999-deadbeef-a2.sending")) == 2
|
||||
|
||||
|
||||
def test_a_torn_claim_is_quarantined_not_deleted(telemetry):
|
||||
"""A non-empty file that parses to nothing is the remainder, not garbage."""
|
||||
telemetry.record("search")
|
||||
claim = telemetry._claim_spool()
|
||||
claim.write_bytes(b"\xff\xfe not utf-8 at all")
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(claim, (stale, stale))
|
||||
|
||||
sent = telemetry.flush()
|
||||
|
||||
assert sent == 0
|
||||
assert not claim.exists()
|
||||
quarantined = list(telemetry.memory_core.data_dir().glob("*.corrupt"))
|
||||
assert len(quarantined) == 1, "torn claim was destroyed instead of kept"
|
||||
|
||||
|
||||
def test_a_failed_rewrite_stops_instead_of_redelivering(telemetry):
|
||||
"""Ignoring the rewrite result reintroduced the duplicates this PR fixes."""
|
||||
for index in range(250):
|
||||
telemetry.record("search", index=index)
|
||||
|
||||
telemetry._rewrite_claim = lambda claim, remaining: False
|
||||
delivered = []
|
||||
telemetry._post = lambda payload, url: delivered.extend(payload.get("batch", [])) or True
|
||||
|
||||
telemetry.flush()
|
||||
assert len(delivered) == 100, f"kept going after a failed rewrite: {len(delivered)}"
|
||||
|
||||
|
||||
def test_partial_files_are_swept(telemetry):
|
||||
"""Nothing else globs *.partial, so a crash mid-rename orphans one forever."""
|
||||
data_dir = telemetry.memory_core.data_dir()
|
||||
data_dir.mkdir(parents=True, exist_ok=True)
|
||||
debris = data_dir / "telemetry-1-abc-a0.1.partial"
|
||||
debris.write_text("x", encoding="utf-8")
|
||||
old = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(debris, (old, old))
|
||||
|
||||
telemetry.flush()
|
||||
assert not debris.exists()
|
||||
|
||||
|
||||
def test_quarantined_batches_are_eventually_collected(telemetry):
|
||||
"""Nothing re-globs .corrupt, so without a sweep they live on disk forever.
|
||||
|
||||
Kept much longer than .partial debris on purpose: a quarantined batch is the
|
||||
only remaining evidence of events that could not be delivered.
|
||||
"""
|
||||
directory = telemetry.memory_core.data_dir()
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
fresh = directory / "telemetry-1-aaaaaaaa-a0.corrupt"
|
||||
old = directory / "telemetry-2-bbbbbbbb-a0.corrupt"
|
||||
for path in (fresh, old):
|
||||
path.write_text("torn", encoding="utf-8")
|
||||
expired = time.time() - (telemetry.CLAIM_EXPIRY_SECONDS + 60)
|
||||
os.utime(old, (expired, expired))
|
||||
|
||||
telemetry._sweep_debris(directory)
|
||||
|
||||
assert fresh.exists(), "a recent quarantine was discarded before anyone could look at it"
|
||||
assert not old.exists(), "an expired quarantine was left on disk forever"
|
||||
|
||||
|
||||
def test_temp_files_orphaned_by_a_kill_are_collected(telemetry):
|
||||
"""_write_identity and _install_salt unlink in a finally, which SIGKILL skips."""
|
||||
directory = telemetry.memory_core.data_dir()
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
orphan = directory / "telemetry-salt.999.tmp"
|
||||
orphan.write_text("abandoned", encoding="utf-8")
|
||||
stale = time.time() - (telemetry.CLAIM_STALE_SECONDS + 60)
|
||||
os.utime(orphan, (stale, stale))
|
||||
|
||||
telemetry._sweep_debris(directory)
|
||||
|
||||
assert not orphan.exists(), "a killed process left a temp file on disk forever"
|
||||
@@ -0,0 +1,274 @@
|
||||
"""Core telemetry behaviour with NO telemetry.init(), in a real subprocess.
|
||||
|
||||
Why this file exists
|
||||
--------------------
|
||||
``telemetry.py`` lives in ``agent-plugin-core/python/`` but its only tests lived
|
||||
under ``claude-code-plugin/tests/``, behind a ``conftest.py`` that calls
|
||||
``configure_harness()`` and ``telemetry.init()`` at import. Core behaviour was
|
||||
therefore only ever exercised inside an already-configured module.
|
||||
|
||||
Two processes in the real pipeline never call ``init()``:
|
||||
|
||||
- ``mcp_server.py``, which records every manual search;
|
||||
- the detached ``python3 telemetry.py`` sender that ``spawn_flush()`` starts at
|
||||
session start, after every skill command, and when the MCP server exits.
|
||||
|
||||
Both fell back to module defaults, so MCP searches reported ``harness=generic``
|
||||
and everything that sender delivered was labelled ``MEM0_PLUGIN`` regardless of
|
||||
which of the six plugins produced it. The suite stayed green throughout.
|
||||
|
||||
These tests run in a fresh interpreter with no conftest, against a built host
|
||||
bundle, which is the only arrangement that can catch that class of bug.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
CORE_ROOT = Path(__file__).resolve().parents[1]
|
||||
REPOSITORY_ROOT = CORE_ROOT.parents[1]
|
||||
HOSTS = {
|
||||
"claude-code": ("claude-code-plugin", "CLAUDE_CODE_PLUGIN"),
|
||||
"cursor": ("cursor-plugin", "CURSOR_PLUGIN"),
|
||||
"codex": ("codex-plugin", "CODEX_PLUGIN"),
|
||||
"kimi": ("kimi-plugin", "KIMI_PLUGIN"),
|
||||
"antigravity": ("antigravity-plugin", "ANTIGRAVITY_PLUGIN"),
|
||||
# Portable: no flush_worker and no hook_runner, so its ONLY sender is the
|
||||
# uninitialised telemetry.py. A native-only test passes here vacuously.
|
||||
"coding-agent": ("mem0-agent-plugin", "CODING_AGENT_PLUGIN"),
|
||||
}
|
||||
|
||||
|
||||
def _core_dir(directory: str) -> Path:
|
||||
return REPOSITORY_ROOT / "integrations" / directory / "core"
|
||||
|
||||
|
||||
def _run(core: Path, data_dir: Path, body: str) -> str:
|
||||
"""Execute `body` in a fresh interpreter with only the host's core on sys.path."""
|
||||
script = f"import sys; sys.path.insert(0, {str(core)!r})\n{body}"
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-c", script],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
env={
|
||||
"MEM0_CODE_DATA_DIR": str(data_dir),
|
||||
"PATH": "/usr/bin:/bin",
|
||||
"HOME": str(data_dir),
|
||||
},
|
||||
)
|
||||
assert result.returncode == 0, result.stderr
|
||||
return result.stdout.strip()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("harness,spec", sorted(HOSTS.items()))
|
||||
def test_identity_resolves_without_init(harness, spec):
|
||||
"""Every built host knows what it is with no configuration call at all."""
|
||||
directory, source_tag = spec
|
||||
core = _core_dir(directory)
|
||||
if not core.exists():
|
||||
pytest.skip(f"{directory} is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
out = _run(
|
||||
core,
|
||||
Path(tmp),
|
||||
"import telemetry; print(telemetry._harness, telemetry._source_tag)",
|
||||
)
|
||||
assert out == f"{harness} {source_tag}"
|
||||
|
||||
|
||||
def test_mcp_server_records_the_real_harness():
|
||||
"""mcp_server imports telemetry and never initialises it (server.py has no init).
|
||||
|
||||
Its recorded events used to carry harness=generic for every plugin.
|
||||
"""
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
data_dir = Path(tmp)
|
||||
_run(
|
||||
core,
|
||||
data_dir,
|
||||
"import mcp_server, telemetry; telemetry.record('search', trigger='mcp-search')",
|
||||
)
|
||||
spooled = (data_dir / "telemetry.jsonl").read_text(encoding="utf-8").strip()
|
||||
|
||||
event = json.loads(spooled)
|
||||
assert event["properties"]["harness"] == "claude-code"
|
||||
assert event["properties"]["source"] == "CLAUDE_CODE_PLUGIN"
|
||||
|
||||
|
||||
def test_the_detached_sender_does_not_relabel_events():
|
||||
"""`python3 telemetry.py` is the sender spawn_flush() starts, and never inits.
|
||||
|
||||
source is stamped at record time now, so which process sends is irrelevant.
|
||||
"""
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
data_dir = Path(tmp)
|
||||
_run(core, data_dir, "import telemetry; telemetry.record('search')")
|
||||
|
||||
captured = data_dir / "captured.json"
|
||||
# Drain with a fresh, unconfigured interpreter, capturing the payload
|
||||
# instead of posting it.
|
||||
_run(
|
||||
core,
|
||||
data_dir,
|
||||
"import json, telemetry\n"
|
||||
"sent = []\n"
|
||||
"telemetry._post = lambda payload, url: sent.append(payload) or True\n"
|
||||
"telemetry.flush()\n"
|
||||
f"open({str(captured)!r}, 'w').write(json.dumps(sent))",
|
||||
)
|
||||
payloads = json.loads(captured.read_text(encoding="utf-8"))
|
||||
|
||||
batches = [p for p in payloads if "batch" in p]
|
||||
assert batches, "nothing was sent"
|
||||
properties = batches[0]["batch"][0]["properties"]
|
||||
assert properties["source"] == "CLAUDE_CODE_PLUGIN"
|
||||
assert properties["harness"] == "claude-code"
|
||||
|
||||
|
||||
def test_every_event_carries_a_uuid_for_dedupe():
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
data_dir = Path(tmp)
|
||||
_run(core, data_dir, "import telemetry; telemetry.record('search'); telemetry.record('flush')")
|
||||
lines = (data_dir / "telemetry.jsonl").read_text(encoding="utf-8").strip().splitlines()
|
||||
|
||||
ids = [json.loads(line)["uuid"] for line in lines]
|
||||
assert len(ids) == 2
|
||||
assert len(set(ids)) == 2
|
||||
|
||||
|
||||
def test_source_tag_defaults_agree_between_the_two_modules():
|
||||
"""configure_harness and telemetry.init must derive the same tag.
|
||||
|
||||
They disagreed: `<host>_plugin` in one and `MEM0_<HOST>_PLUGIN` in the other,
|
||||
so one plugin could emit three different source values depending on which
|
||||
process sent the batch.
|
||||
"""
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
out = _run(
|
||||
core,
|
||||
Path(tmp),
|
||||
"import memory_core, telemetry\n"
|
||||
"memory_core.configure_harness('kimi')\n"
|
||||
"telemetry.init(harness='kimi')\n"
|
||||
"print(memory_core.harness_config()['source_tag'].upper(), telemetry._source_tag)",
|
||||
)
|
||||
left, right = out.split()
|
||||
assert left == right == "KIMI_PLUGIN"
|
||||
|
||||
|
||||
def test_the_plugin_declares_its_surface_in_the_body_and_the_headers():
|
||||
"""Body and headers both, because only the body works on every backend."""
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
out = _run(
|
||||
core,
|
||||
Path(tmp),
|
||||
"import json, memory_core\n"
|
||||
"h = memory_core.platform_headers('k')\n"
|
||||
"print(json.dumps({'source': h.get('X-Mem0-Source'),"
|
||||
" 'app': h.get('X-Application'),"
|
||||
" 'client': h.get('X-Mem0-Client'),"
|
||||
" 'auth': h.get('Authorization'),"
|
||||
" 'ctype': h.get('Content-Type')}))",
|
||||
)
|
||||
headers = json.loads(out)
|
||||
assert headers["source"] == "MEM0_PLUGIN"
|
||||
assert headers["app"] == "claude-code"
|
||||
assert headers["client"].startswith("mem0-plugin/")
|
||||
# The transport headers the three call sites relied on must survive.
|
||||
assert headers["auth"] == "Token k"
|
||||
assert headers["ctype"] == "application/json"
|
||||
|
||||
|
||||
def _session_start(core: Path, data_dir: Path) -> list[str]:
|
||||
"""Drive the real hook_runner session-start path and return lifecycle events."""
|
||||
recorded = "\n".join(
|
||||
[
|
||||
"import io, json, sys",
|
||||
f"sys.path.insert(0, {str(core)!r})",
|
||||
"import telemetry, hook_runner",
|
||||
"seen = []",
|
||||
"telemetry.record = lambda event, **kw: seen.append(event) or None",
|
||||
"telemetry.spawn_flush = lambda: False",
|
||||
# run() reads sys.argv through argparse; it takes no positional args.
|
||||
"sys.argv = ['hook_runner', 'session-start']",
|
||||
"sys.stdin = io.StringIO('{}')",
|
||||
"hook_runner.run()",
|
||||
"print(json.dumps([e for e in seen if e in ('install', 'upgrade')]))",
|
||||
]
|
||||
)
|
||||
import json as _json
|
||||
|
||||
return _json.loads(_run(core, data_dir, recorded) or "[]")
|
||||
|
||||
|
||||
def test_a_fresh_install_reports_install_not_upgrade():
|
||||
"""The decision must survive the writes hook_runner does before asking.
|
||||
|
||||
claim_install() is reached only after cache_plugin_api_key() has written
|
||||
`api-key` and EvidenceStore() has created `evidence.sqlite3`. Asking "is the
|
||||
data dir empty" at that point always saw content, so code.install could
|
||||
never fire and every new user was counted as an upgrade.
|
||||
"""
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
data_dir = Path(tmp) / "data"
|
||||
assert _session_start(core, data_dir) == ["install"]
|
||||
|
||||
|
||||
def test_the_lifecycle_event_fires_exactly_once():
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
data_dir = Path(tmp) / "data"
|
||||
first = _session_start(core, data_dir)
|
||||
second = _session_start(core, data_dir)
|
||||
third = _session_start(core, data_dir)
|
||||
|
||||
assert first == ["install"]
|
||||
assert second == []
|
||||
assert third == []
|
||||
|
||||
|
||||
def test_an_existing_data_dir_reports_upgrade():
|
||||
core = _core_dir("claude-code-plugin")
|
||||
if not core.exists():
|
||||
pytest.skip("claude-code-plugin is not built in this tree")
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
data_dir = Path(tmp) / "data"
|
||||
data_dir.mkdir(parents=True)
|
||||
# A 0.2.x leftover: the data dir survives the upgrade.
|
||||
(data_dir / "requirements.txt").write_text("mem0ai\n", encoding="utf-8")
|
||||
assert _session_start(core, data_dir) == ["upgrade"]
|
||||
@@ -1,3 +1,5 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
|
||||
import { redactSecrets } from "./lifecycle.ts";
|
||||
|
||||
const POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX";
|
||||
@@ -74,34 +76,108 @@ export function errorKind(error: unknown): string {
|
||||
return error instanceof Error ? error.constructor.name : "other";
|
||||
}
|
||||
|
||||
// Delivery is retried in memory, not spooled to disk, and that is a decision
|
||||
// rather than an omission. The Python core spools because its hooks are separate
|
||||
// processes that fire per tool call and exit immediately, so nothing survives
|
||||
// without a file. These plugins are loaded into a host that lives for a whole
|
||||
// session, so re-queueing covers the same transient failures without the claim
|
||||
// and lease machinery a correct cross-process spool needs. What that leaves
|
||||
// uncovered is narrow: a session that both starts and ends with no connectivity.
|
||||
const RETRY_BACKOFF_CEILING_MS = 60_000;
|
||||
// Consecutive failed flushes before the queue is dropped. Deliberately NOT the
|
||||
// same thing as Python's budget, which rides in the claim filename and so
|
||||
// follows one batch: this counter lives in the closure and counts the outage,
|
||||
// not the payload. Events captured between attempts join the same queue and go
|
||||
// with it. Per-batch accounting would need an attempt count on every event, and
|
||||
// the queue is already bounded, so the simpler rule is the one in force here.
|
||||
// Without any bound a payload the server will never accept is retried for the
|
||||
// whole session and, now that the backlog is preferred over new events, holds
|
||||
// the queue against everything behind it.
|
||||
const MAX_DELIVERY_ATTEMPTS = 5;
|
||||
|
||||
export function createTelemetry(config: TelemetryConfig) {
|
||||
let queue: Record<string, unknown>[] = [];
|
||||
let timer: ReturnType<typeof setInterval> | undefined;
|
||||
let consecutiveFailures = 0;
|
||||
let retryNotBefore = 0;
|
||||
let exitFlushAttempted = false;
|
||||
let flushing = false;
|
||||
const flushThreshold = config.flushThreshold ?? 10;
|
||||
const maxQueueSize = config.maxQueueSize ?? 100;
|
||||
|
||||
const deliver = config.delivery ?? (async (batch: Record<string, unknown>[]) => {
|
||||
await fetch(POSTHOG_BATCH_URL, {
|
||||
const response = await fetch(POSTHOG_BATCH_URL, {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ api_key: POSTHOG_API_KEY, batch }),
|
||||
signal: AbortSignal.timeout(3_000),
|
||||
});
|
||||
// fetch only rejects on a network-level failure. Without this check a 500,
|
||||
// a 503 or a 429 resolved normally and the batch was counted as delivered
|
||||
// and dropped, which is the likelier outage than a refused connection.
|
||||
// Any non-2xx is retried, matching the Python core: the backoff and the
|
||||
// queue bound contain a payload that will never be accepted, because the
|
||||
// re-queued batch sits at the front and is the first thing evicted.
|
||||
if (!response.ok) throw new Error(`posthog responded ${response.status}`);
|
||||
});
|
||||
|
||||
async function flush(): Promise<void> {
|
||||
async function flush(force = false): Promise<void> {
|
||||
// One at a time. Two overlapping flushes each detach the queue and each
|
||||
// prepend their own batch back on failure, so the later batch lands in front
|
||||
// of the earlier one and the truncation then drops the OLDER events first,
|
||||
// inverting the priority the failure path exists to establish. A second
|
||||
// caller returns immediately; the queue waits for the next flush.
|
||||
if (flushing) return;
|
||||
if (!queue.length) return;
|
||||
// `force` skips the cooldown. beforeExit is the last chance this process
|
||||
// gets, and gating it on the same backoff meant that after any failure the
|
||||
// exit flush did nothing and the queue died with the process, which is the
|
||||
// loss this whole mechanism exists to prevent.
|
||||
if (!force && Date.now() < retryNotBefore) return;
|
||||
const batch = queue;
|
||||
queue = [];
|
||||
flushing = true;
|
||||
try {
|
||||
await deliver(batch);
|
||||
consecutiveFailures = 0;
|
||||
retryNotBefore = 0;
|
||||
} catch {
|
||||
// Telemetry must never affect plugin behavior.
|
||||
// Put it back. Detaching the batch and swallowing the error deleted the
|
||||
// events outright, so any blip silently dropped telemetry with nothing
|
||||
// recording that it had happened. Every event carries a uuid, so a retry
|
||||
// that duplicates one PostHog already accepted is collapsed there.
|
||||
//
|
||||
consecutiveFailures += 1;
|
||||
if (consecutiveFailures >= MAX_DELIVERY_ATTEMPTS) {
|
||||
// Give up on the queue so a failing outage cannot hold it for the
|
||||
// session. This drops whatever is queued now, which includes events
|
||||
// captured during the outage, not only the batch that kept failing.
|
||||
consecutiveFailures = 0;
|
||||
retryNotBefore = 0;
|
||||
return;
|
||||
}
|
||||
// Keep the FRONT on overflow, so the batch being retried survives and a
|
||||
// new event is what gets dropped. Matches the Python core, where record()
|
||||
// refuses new events once the spool is full rather than evicting the
|
||||
// backlog. Keeping the newest would throw away exactly the events this
|
||||
// retry exists to save.
|
||||
queue = [...batch, ...queue].slice(0, maxQueueSize);
|
||||
retryNotBefore = Date.now() + Math.min(2 ** consecutiveFailures * 1_000, RETRY_BACKOFF_CEILING_MS);
|
||||
} finally {
|
||||
flushing = false;
|
||||
}
|
||||
}
|
||||
|
||||
function beforeExit(): void {
|
||||
void flush();
|
||||
// Once, and only once. Node re-emits beforeExit whenever the handler
|
||||
// schedules more async work, so an unconditional forced flush looped until
|
||||
// the attempt budget was spent: five attempts against a 3s delivery timeout
|
||||
// is fifteen seconds added to the shutdown of whatever editor or CLI is
|
||||
// hosting this. The backoff used to end that loop after one attempt, and
|
||||
// removing it for the forced path removed the only thing bounding it.
|
||||
if (exitFlushAttempted) return;
|
||||
exitFlushAttempted = true;
|
||||
void flush(true);
|
||||
}
|
||||
|
||||
function build(event: string, properties: Record<string, unknown> = {}): Record<string, unknown> | null {
|
||||
@@ -112,6 +188,15 @@ export function createTelemetry(config: TelemetryConfig) {
|
||||
return {
|
||||
event: config.eventName?.(event) ?? event,
|
||||
distinct_id: distinctId,
|
||||
// Stamped once, at capture. This is what makes retrying safe: a batch
|
||||
// re-sent after a failure carries the same ids, so PostHog collapses
|
||||
// anything it already accepted instead of counting it twice.
|
||||
uuid: randomUUID(),
|
||||
// Capture time, not ingestion time. Events now sit through backoff and
|
||||
// across a whole outage, so without this PostHog records them whenever
|
||||
// delivery happened to succeed. It also matters for the uuid dedupe
|
||||
// above, whose key includes the event date.
|
||||
timestamp: new Date().toISOString(),
|
||||
properties: {
|
||||
...safeProperties(properties),
|
||||
...safeProperties(config.commonProperties ?? {}),
|
||||
@@ -134,8 +219,10 @@ export function createTelemetry(config: TelemetryConfig) {
|
||||
try {
|
||||
const payload = build(event, properties);
|
||||
if (!payload) return;
|
||||
// Full means drop this event, not evict the backlog. Same rule as the
|
||||
// failure path above and as Python's record().
|
||||
if (queue.length >= maxQueueSize) return;
|
||||
queue.push(payload);
|
||||
if (queue.length > maxQueueSize) queue = queue.slice(-maxQueueSize);
|
||||
if (!timer) {
|
||||
timer = setInterval(() => void flush(), config.flushIntervalMs ?? 5_000);
|
||||
timer.unref?.();
|
||||
@@ -149,6 +236,8 @@ export function createTelemetry(config: TelemetryConfig) {
|
||||
|
||||
function resetForTesting(): void {
|
||||
queue = [];
|
||||
consecutiveFailures = 0;
|
||||
retryNotBefore = 0;
|
||||
if (timer) clearInterval(timer);
|
||||
timer = undefined;
|
||||
process.off("beforeExit", beforeExit);
|
||||
|
||||
@@ -122,3 +122,250 @@ test("error classification does not expose messages", () => {
|
||||
assert.equal(errorKind(new Error("request timeout")), "timeout");
|
||||
assert.equal(errorKind(new Error("fetch failed")), "network");
|
||||
});
|
||||
|
||||
test("a failed delivery keeps the batch instead of deleting it", async () => {
|
||||
// The defect: the queue was detached before the await and the error swallowed,
|
||||
// so one blip destroyed the events with nothing recording that it happened.
|
||||
const attempts: Record<string, unknown>[][] = [];
|
||||
let failNext = true;
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d",
|
||||
flushThreshold: 1000,
|
||||
delivery: async (batch) => {
|
||||
attempts.push(batch);
|
||||
if (failNext) throw new Error("network down");
|
||||
},
|
||||
});
|
||||
|
||||
telemetry.capture("one");
|
||||
telemetry.capture("two");
|
||||
await telemetry.flush();
|
||||
|
||||
assert.equal(attempts.length, 1);
|
||||
assert.equal(telemetry.queueForTesting().length, 2, "events were dropped on failure");
|
||||
|
||||
failNext = false;
|
||||
// Backoff is in force, so wait it out the way wall time would.
|
||||
await new Promise((resolve) => setTimeout(resolve, 2_100));
|
||||
await telemetry.flush();
|
||||
|
||||
assert.equal(attempts.length, 2, "never retried");
|
||||
assert.equal(telemetry.queueForTesting().length, 0);
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("a retried event carries the same uuid so PostHog can collapse it", async () => {
|
||||
const attempts: Record<string, unknown>[][] = [];
|
||||
let failNext = true;
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d",
|
||||
flushThreshold: 1000,
|
||||
delivery: async (batch) => {
|
||||
attempts.push(batch);
|
||||
if (failNext) throw new Error("network down");
|
||||
},
|
||||
});
|
||||
|
||||
telemetry.capture("once");
|
||||
await telemetry.flush();
|
||||
failNext = false;
|
||||
await new Promise((resolve) => setTimeout(resolve, 2_100));
|
||||
await telemetry.flush();
|
||||
|
||||
assert.equal(attempts.length, 2);
|
||||
const first = attempts[0][0].uuid;
|
||||
assert.ok(first, "events carry no uuid, so a retry would double count");
|
||||
assert.equal(attempts[1][0].uuid, first, "retry minted a new uuid");
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("repeated failures back off instead of retrying every flush", async () => {
|
||||
let calls = 0;
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d",
|
||||
flushThreshold: 1000,
|
||||
delivery: async () => { calls += 1; throw new Error("blocked"); },
|
||||
});
|
||||
|
||||
telemetry.capture("one");
|
||||
await telemetry.flush();
|
||||
await telemetry.flush();
|
||||
await telemetry.flush();
|
||||
|
||||
assert.equal(calls, 1, "a blocked host was hammered on every flush");
|
||||
assert.equal(telemetry.queueForTesting().length, 1, "the event was lost while backing off");
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("a full queue drops the new event and keeps the batch being retried", async () => {
|
||||
// Python's record() refuses new events once the spool is full rather than
|
||||
// evicting the backlog. Keeping the newest here would throw away exactly the
|
||||
// events the retry exists to save.
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d",
|
||||
flushThreshold: 1000, maxQueueSize: 3,
|
||||
delivery: async () => { throw new Error("down"); },
|
||||
});
|
||||
|
||||
// Fill past the cap BEFORE the flush, so the re-queue actually has to truncate.
|
||||
// Capturing only two left the queue empty at re-queue time and the slice on the
|
||||
// failure path never ran, which is the half that decides the direction.
|
||||
telemetry.capture("a");
|
||||
telemetry.capture("b");
|
||||
telemetry.capture("c");
|
||||
await telemetry.flush();
|
||||
telemetry.capture("d");
|
||||
telemetry.capture("e");
|
||||
|
||||
const events = telemetry.queueForTesting().map((e) => (e as any).event);
|
||||
assert.equal(events.length, 3, "queue grew past maxQueueSize");
|
||||
assert.deepEqual(events, ["a", "b", "c"], "the retried batch was evicted instead of the new events");
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("the exit-time flush ignores the backoff", async () => {
|
||||
// beforeExit is the last chance the process gets. Gating it on the same
|
||||
// cooldown meant that after any failure it did nothing and the queue died.
|
||||
let attempts = 0;
|
||||
let failing = true;
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d", flushThreshold: 1000,
|
||||
delivery: async () => { attempts += 1; if (failing) throw new Error("down"); },
|
||||
});
|
||||
|
||||
telemetry.capture("a");
|
||||
await telemetry.flush();
|
||||
assert.equal(attempts, 1);
|
||||
|
||||
failing = false;
|
||||
await telemetry.flush();
|
||||
assert.equal(attempts, 1, "the backoff should still hold for an ordinary flush");
|
||||
|
||||
await telemetry.flush(true);
|
||||
assert.equal(attempts, 2, "the exit flush was suppressed by the backoff");
|
||||
assert.equal(telemetry.queueForTesting().length, 0);
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("every event carries a capture-time timestamp", async () => {
|
||||
const sent: Record<string, unknown>[][] = [];
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d", flushThreshold: 1000,
|
||||
delivery: async (batch) => { sent.push(batch); },
|
||||
});
|
||||
|
||||
telemetry.capture("a");
|
||||
const capturedAt = Date.now();
|
||||
await new Promise((resolve) => setTimeout(resolve, 50));
|
||||
await telemetry.flush();
|
||||
|
||||
const stamped = sent[0][0].timestamp as string;
|
||||
assert.ok(stamped, "no timestamp, so PostHog would record delivery time");
|
||||
assert.ok(Math.abs(Date.parse(stamped) - capturedAt) < 1_000, "not capture time");
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("a batch the server will never accept is eventually given up on", async () => {
|
||||
let attempts = 0;
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d", flushThreshold: 1000,
|
||||
delivery: async () => { attempts += 1; throw new Error("permanently bad"); },
|
||||
});
|
||||
|
||||
telemetry.capture("doomed");
|
||||
for (let i = 0; i < 8; i += 1) await telemetry.flush(true);
|
||||
|
||||
assert.ok(attempts <= 6, `retried ${attempts} times with no cap`);
|
||||
assert.equal(telemetry.queueForTesting().length, 0, "a doomed batch held the queue forever");
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("an HTTP error response is a failure, not a delivery", async () => {
|
||||
// fetch only rejects on a network-level failure, so a 500 used to resolve
|
||||
// normally and the batch was dropped as delivered. Exercises the real default
|
||||
// delivery path rather than an injected one, which is where this hid.
|
||||
const realFetch = globalThis.fetch;
|
||||
let calls = 0;
|
||||
globalThis.fetch = (async () => {
|
||||
calls += 1;
|
||||
return new Response("upstream is unwell", { status: 503 });
|
||||
}) as typeof fetch;
|
||||
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d", flushThreshold: 1000,
|
||||
});
|
||||
try {
|
||||
telemetry.capture("during.outage");
|
||||
await telemetry.flush();
|
||||
|
||||
assert.equal(calls, 1, "never reached the network");
|
||||
assert.equal(telemetry.queueForTesting().length, 1, "a 503 was counted as delivered");
|
||||
} finally {
|
||||
globalThis.fetch = realFetch;
|
||||
telemetry.resetForTesting();
|
||||
}
|
||||
});
|
||||
|
||||
test("a 2xx is a delivery", async () => {
|
||||
const realFetch = globalThis.fetch;
|
||||
globalThis.fetch = (async () => new Response("ok", { status: 200 })) as typeof fetch;
|
||||
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d", flushThreshold: 1000,
|
||||
});
|
||||
try {
|
||||
telemetry.capture("fine");
|
||||
await telemetry.flush();
|
||||
assert.equal(telemetry.queueForTesting().length, 0, "a good response did not clear the queue");
|
||||
} finally {
|
||||
globalThis.fetch = realFetch;
|
||||
telemetry.resetForTesting();
|
||||
}
|
||||
});
|
||||
|
||||
test("the exit flush is attempted once, not until the budget is spent", async () => {
|
||||
// Node re-emits beforeExit whenever the handler schedules async work, so an
|
||||
// unconditional forced flush looped until MAX_DELIVERY_ATTEMPTS. Against the
|
||||
// real 3s delivery timeout that is fifteen seconds added to a host's shutdown.
|
||||
let attempts = 0;
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d", flushThreshold: 1000,
|
||||
delivery: async () => { attempts += 1; throw new Error("down"); },
|
||||
});
|
||||
|
||||
telemetry.capture("a");
|
||||
const handlers = process.listeners("beforeExit");
|
||||
const ours = handlers[handlers.length - 1] as () => void;
|
||||
ours();
|
||||
ours();
|
||||
ours();
|
||||
await new Promise((resolve) => setTimeout(resolve, 20));
|
||||
|
||||
assert.equal(attempts, 1, `exit flush ran ${attempts} times`);
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
test("overlapping flushes do not reorder the backlog behind newer events", async () => {
|
||||
// Each flush detaches the queue and prepends its own batch back on failure, so
|
||||
// two in flight at once put the LATER batch in front of the earlier one. The
|
||||
// truncation then drops the older events first, inverting the priority the
|
||||
// failure path exists to establish.
|
||||
let release: (() => void)[] = [];
|
||||
const telemetry = createTelemetry({
|
||||
host: "h", source: "S", version: "1", distinctId: "d", flushThreshold: 1000,
|
||||
delivery: () => new Promise((_resolve, reject) => { release.push(() => reject(new Error("down"))); }),
|
||||
});
|
||||
|
||||
telemetry.capture("first");
|
||||
const a = telemetry.flush();
|
||||
telemetry.capture("second");
|
||||
const b = telemetry.flush();
|
||||
|
||||
release.forEach((fn) => fn());
|
||||
await Promise.all([a, b]);
|
||||
|
||||
const events = telemetry.queueForTesting().map((e) => (e as any).event);
|
||||
assert.equal(release.length, 1, "a second delivery started while one was in flight");
|
||||
assert.deepEqual(events, ["first", "second"], `backlog reordered: ${events.join(",")}`);
|
||||
telemetry.resetForTesting();
|
||||
});
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Generated by integrations/agent-plugin-core/build/build.py. Do not edit."""
|
||||
|
||||
HARNESS_ID = "antigravity"
|
||||
SOURCE_TAG = "ANTIGRAVITY_PLUGIN"
|
||||
|
||||
# Platform-side vocabulary (mem0_event.source + X-Application). The whole
|
||||
# plugin family is one source; which editor it runs in is the application.
|
||||
# An empty application means the host is unknown, and memory_core omits
|
||||
# the header entirely rather than sending a placeholder.
|
||||
PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
PLATFORM_APPLICATION = "antigravity"
|
||||
@@ -290,6 +290,11 @@ def run(
|
||||
if args.plugin_data_dir:
|
||||
os.environ[data_dir_env] = args.plugin_data_dir
|
||||
|
||||
# Snapshot BEFORE anything writes to the data dir: cache_plugin_api_key
|
||||
# writes `api-key` and EvidenceStore creates `evidence.sqlite3`, so asking
|
||||
# after them always saw content and every fresh install reported an upgrade.
|
||||
data_dir_was_empty = telemetry.data_dir_was_empty()
|
||||
|
||||
cache_plugin_api_key()
|
||||
if args.action == "session-start":
|
||||
clear_stale_api_key_cache()
|
||||
@@ -305,8 +310,19 @@ def run(
|
||||
return 0
|
||||
|
||||
if args.action == "session-start":
|
||||
if telemetry.is_first_run():
|
||||
# Claims the marker atomically and says which event to record, so a
|
||||
# second session starting alongside this one cannot record it too.
|
||||
first_event = telemetry.claim_install(was_empty=data_dir_was_empty)
|
||||
if first_event == "install":
|
||||
telemetry.record("install")
|
||||
elif first_event == "upgrade":
|
||||
# First run after a build that never wrote the marker; the
|
||||
# predecessor version was never recorded anywhere.
|
||||
telemetry.record("upgrade", from_version="pre-0.3")
|
||||
else:
|
||||
previous = telemetry.claim_version_change()
|
||||
if previous:
|
||||
telemetry.record("upgrade", from_version=previous)
|
||||
recovered = recover_pending_handoffs()
|
||||
record_session_start(store, hook_input)
|
||||
if recovered:
|
||||
|
||||
@@ -29,7 +29,7 @@ from typing import Any, Iterable
|
||||
import telemetry
|
||||
|
||||
DEFAULT_API_URL = "https://api.mem0.ai"
|
||||
PLUGIN_VERSION = "0.3.1"
|
||||
PLUGIN_VERSION = "0.3.2"
|
||||
|
||||
_harness_name: str = "generic"
|
||||
_harness_env_prefix: str = "MEM0_PLUGIN"
|
||||
@@ -1800,6 +1800,34 @@ def extraction_message_batches(
|
||||
return batches
|
||||
|
||||
|
||||
# Platform surface attribution. Read from the generated per-host module so a new
|
||||
# entrypoint is correct without remembering to configure anything.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
except ImportError:
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
|
||||
def platform_headers(key: str) -> dict[str, str]:
|
||||
"""Auth plus the three surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are set-once by contract: this is the
|
||||
outermost layer, so it sets them, and nothing below may overwrite them.
|
||||
X-Mem0-Client is append-only — anything downstream adds itself to the tail.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": _PLATFORM_SOURCE,
|
||||
"X-Mem0-Client": f"mem0-plugin/{PLUGIN_VERSION}",
|
||||
}
|
||||
if _PLATFORM_APPLICATION:
|
||||
headers["X-Application"] = _PLATFORM_APPLICATION
|
||||
return headers
|
||||
|
||||
|
||||
def _request_json(
|
||||
url: str, key: str, payload: dict[str, Any], timeout: float
|
||||
) -> tuple[dict[str, Any] | list[Any], int, int]:
|
||||
@@ -1807,7 +1835,7 @@ def _request_json(
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1834,7 +1862,7 @@ def _get_json(
|
||||
) -> tuple[dict[str, Any] | list[Any], int]:
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="GET",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1980,6 +2008,13 @@ def flush_session(
|
||||
"user_id": write_user,
|
||||
"app_id": repo.app_id,
|
||||
"run_id": session_id,
|
||||
# Top level, not metadata: the backend reads `source` from the body or
|
||||
# the query string, never from metadata, which is where this used to
|
||||
# sit. The X-Mem0-Source header is also read, but only from the
|
||||
# platform release that ships alongside this change, so the body value
|
||||
# is what makes attribution work on both. The harness tag stays in
|
||||
# metadata as hook provenance.
|
||||
"source": _PLATFORM_SOURCE,
|
||||
"metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)},
|
||||
"agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS,
|
||||
"custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS,
|
||||
@@ -2523,7 +2558,7 @@ def _collect_memory_ids(
|
||||
def _delete_memory(api_url: str, key: str, memory_id: str) -> bool:
|
||||
request = urllib.request.Request(
|
||||
f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/",
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Anonymous usage telemetry for Mem0 agent plugins.
|
||||
"""Usage telemetry for Mem0 agent plugins.
|
||||
|
||||
Events are linked to your Mem0 account email when an API key is configured, and
|
||||
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
||||
attribute the same way.
|
||||
|
||||
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
||||
touches the network: `record` appends one JSON line to a local spool and returns.
|
||||
@@ -9,7 +13,8 @@ started once per session and again from the flush worker that is already detache
|
||||
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
||||
|
||||
Never sends prompts, memory text, queries, file paths, repository names, or API
|
||||
keys: only event names, durations, counts, coarse outcomes, and salted hashes.
|
||||
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
||||
identifiers hashed with a random per-install salt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,8 +34,24 @@ from typing import Any
|
||||
|
||||
import memory_core
|
||||
|
||||
_harness: str = "generic"
|
||||
_source_tag: str = "MEM0_PLUGIN"
|
||||
# Seeded from the per-host module the build generates into core/. Two processes
|
||||
# in this pipeline never call init() — mcp_server.py, and the detached
|
||||
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
||||
# was what every one of their events got labelled with.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
||||
except ImportError:
|
||||
_DEFAULT_HARNESS = "generic"
|
||||
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
_salt_cache: str = ""
|
||||
_harness: str = _DEFAULT_HARNESS
|
||||
_source_tag: str = _DEFAULT_SOURCE_TAG
|
||||
_PRIVATE_KEYS = {
|
||||
"apikey",
|
||||
"authorization",
|
||||
@@ -56,10 +77,19 @@ _PRIVATE_KEYS = {
|
||||
}
|
||||
|
||||
|
||||
def init(harness: str = "generic", source_tag: str = "") -> None:
|
||||
def init(harness: str = "", source_tag: str = "") -> None:
|
||||
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
||||
|
||||
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
||||
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
||||
meant one plugin could emit three different source values depending on which
|
||||
process happened to send the batch.
|
||||
"""
|
||||
global _harness, _source_tag
|
||||
_harness = harness
|
||||
_source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN"
|
||||
_harness = harness or _DEFAULT_HARNESS
|
||||
_source_tag = source_tag or (
|
||||
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
||||
)
|
||||
|
||||
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
||||
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
||||
@@ -70,6 +100,16 @@ BATCH_SIZE = 100
|
||||
SEND_TIMEOUT = 5
|
||||
CLAIM_STALE_SECONDS = 120
|
||||
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
||||
# A batch is only discarded once it has genuinely been retried this many times.
|
||||
MAX_CLAIM_ATTEMPTS = 3
|
||||
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
||||
# cannot turn one flush into an unbounded send loop.
|
||||
MAX_PARKED_PER_RUN = 3
|
||||
# Added to the wait before a released claim becomes reclaimable, per attempt
|
||||
# already spent. Releasing straight to "reclaimable now" let two senders burn the
|
||||
# whole budget within seconds of one another on a single momentary failure, and
|
||||
# discard a batch a retry a minute later would have delivered.
|
||||
RETRY_COOLDOWN_SECONDS = 60
|
||||
|
||||
|
||||
def is_enabled() -> bool:
|
||||
@@ -83,9 +123,126 @@ def is_enabled() -> bool:
|
||||
|
||||
|
||||
def _digest(value: str, length: int = 16) -> str:
|
||||
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _salt_path() -> Path:
|
||||
return memory_core.data_dir() / "telemetry-salt"
|
||||
|
||||
|
||||
def _install_salt() -> str:
|
||||
"""Random per-install salt, created once and memoized for the process.
|
||||
|
||||
Deliberately its own file, claimed with O_CREAT|O_EXCL, rather than a key in
|
||||
the identity file. Three reasons, all of which produced wrong data when this
|
||||
lived in the identity dict:
|
||||
|
||||
- Hooks are short-lived separate processes firing on every tool call, and
|
||||
people run more than one agent window. A read-modify-write would let each
|
||||
process mint its own salt, so one repository would hash several ways in the
|
||||
window before a writer won.
|
||||
- resolve_distinct_id holds a copy of the identity dict across a network call
|
||||
to /v1/ping/, so whichever write landed second erased the other's key —
|
||||
losing either the salt (repo_hash changes mid-stream) or the email (a
|
||||
second $identify, splitting the person).
|
||||
- Touching the identity file from record() would create it, and is_first_run
|
||||
keys off that file, so recording an event would silently suppress the
|
||||
install event.
|
||||
|
||||
Published atomically, and there is deliberately no derived fallback. Creating
|
||||
the file with O_CREAT|O_EXCL and then writing into it leaves a window where
|
||||
the file exists and is empty, and a concurrent hook that reads it in that
|
||||
window gets nothing. Falling back to a digest of the path would hand that
|
||||
process a salt an attacker can compute, memoized for its whole run, which is
|
||||
the privacy control this function exists to provide silently turning itself
|
||||
off under load. The salt is written to a private temp file first and linked
|
||||
into place, so the name either does not exist or already has the full value.
|
||||
|
||||
Returns "" when it genuinely cannot persist. Callers omit the hash entirely
|
||||
rather than emit an unsalted one.
|
||||
"""
|
||||
global _salt_cache
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
|
||||
path = _salt_path()
|
||||
# Read before writing. Hooks are separate processes firing on every tool
|
||||
# call, so all but the first find the salt already published; going straight
|
||||
# to create-fsync-link-unlink meant every one of them paid an fsync to
|
||||
# discover that, on a path whose whole promise is appending a line and
|
||||
# returning.
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
temporary = path.with_name(f"{path.name}.{os.getpid()}.tmp")
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
stream.write(uuid.uuid4().hex)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
try:
|
||||
# Atomic claim: fails if another process already published one.
|
||||
# os.link rather than replace, which would clobber theirs.
|
||||
os.link(temporary, path)
|
||||
except FileExistsError:
|
||||
pass
|
||||
except OSError:
|
||||
# No hardlinks here (some network mounts, some container volumes).
|
||||
# Claim the name directly instead. That reopens the empty-file
|
||||
# window, but the window is now benign: a reader that lands in it
|
||||
# gets "" and omits the hash for that process rather than caching a
|
||||
# guessable one. Losing the hashes on every run of an entire
|
||||
# filesystem is the worse failure.
|
||||
try:
|
||||
fallback = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fallback, "w", encoding="utf-8") as stream:
|
||||
stream.write(temporary.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_salt_cache = ""
|
||||
return _salt_cache
|
||||
|
||||
|
||||
def _scoped_digest(value: str, length: int = 16) -> str:
|
||||
"""Salted digest for values drawn from a guessable space.
|
||||
|
||||
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
||||
no remote — which normally contains the account username. Sixteen unsalted
|
||||
hex characters over that input space is enumerable, so this is not a
|
||||
privacy control without the salt. Salting per install keeps every
|
||||
within-account join the analytics actually use and gives up only
|
||||
cross-machine joins on the same repository, which nothing computes.
|
||||
|
||||
Returns "" when there is no salt, so record() omits the property. An
|
||||
unsalted digest over this input space is close to plaintext, and emitting one
|
||||
under a name that implies it is hashed is worse than sending nothing.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
salt = _install_salt()
|
||||
if not salt:
|
||||
return ""
|
||||
return hashlib.sha256(f"{salt}:{value}".encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _safe_value(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return memory_core.redact(value)
|
||||
@@ -145,9 +302,176 @@ def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
||||
return created
|
||||
|
||||
|
||||
def _rotate_anonymous_id(identity: dict[str, str]) -> str:
|
||||
"""Mint a fresh anonymous id because the account context is gone.
|
||||
|
||||
The previous id may already have been merged into a person profile by an
|
||||
$identify, and that merge is permanent. Reusing it after a logout or a key
|
||||
change attributes everything that follows to the account that just went
|
||||
away, which is the same misattribution the key fingerprint exists to stop,
|
||||
only arriving through the anonymous path instead.
|
||||
|
||||
`aliased` is cleared with it: the new id has never been merged, so it is
|
||||
eligible to be aliased into whatever account comes next.
|
||||
"""
|
||||
created = f"code-anon-{uuid.uuid4().hex}"
|
||||
identity["anonymous_id"] = created
|
||||
identity.pop("aliased", None)
|
||||
_write_identity(identity)
|
||||
return created
|
||||
|
||||
|
||||
def _install_state_path() -> Path:
|
||||
return memory_core.data_dir() / "install-state.json"
|
||||
|
||||
|
||||
def is_first_run() -> bool:
|
||||
"""Whether this machine has never recorded a plugin event before."""
|
||||
return not _identity_path().exists()
|
||||
"""Whether install has never been recorded on this machine.
|
||||
|
||||
Deliberately NOT the identity file. That file is only written by a
|
||||
successful flush, so an offline or firewalled user recorded code.install on
|
||||
every single session, forever — and every 0.2.x user recorded one on their
|
||||
first 0.3.x session because 0.2.x never wrote it at all.
|
||||
"""
|
||||
return not _install_state_path().exists()
|
||||
|
||||
|
||||
def data_dir_was_empty() -> bool:
|
||||
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
||||
|
||||
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
||||
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
||||
asking at claim time always saw content and every fresh install reported an
|
||||
upgrade. The caller snapshots this at the top of the run instead.
|
||||
"""
|
||||
return not _data_dir_has_content()
|
||||
|
||||
|
||||
def claim_install(was_empty: bool | None = None) -> str | None:
|
||||
"""Claim the one install/upgrade record for this machine, atomically.
|
||||
|
||||
Returns the event to record ("install" or "upgrade"), or None if another
|
||||
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
||||
cannot both win.
|
||||
|
||||
`was_empty` must come from data_dir_was_empty() called before this process
|
||||
wrote anything. Omitting it falls back to checking now, which is only
|
||||
correct for a caller that has touched nothing.
|
||||
"""
|
||||
if not is_enabled():
|
||||
# Never consume the one-shot claim while the user is opted out, or they
|
||||
# would silently lose their install event if they later opt in.
|
||||
return None
|
||||
|
||||
path = _install_state_path()
|
||||
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
json.dump(
|
||||
{
|
||||
"plugin_version": memory_core.PLUGIN_VERSION,
|
||||
"installed_at": memory_core.utc_now(),
|
||||
"upgraded": upgrading,
|
||||
},
|
||||
stream,
|
||||
)
|
||||
# Durable before this returns. The O_EXCL open is what makes the
|
||||
# claim exclusive, so it cannot be replaced by a temp-and-rename
|
||||
# without losing that, which leaves the content as the thing to make
|
||||
# safe. A kill between the open and this fsync used to leave a marker
|
||||
# that exists but parses to nothing: is_first_run reads it as claimed
|
||||
# and claim_version_change cannot read a version out of it.
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
except OSError:
|
||||
pass
|
||||
return "upgrade" if upgrading else "install"
|
||||
|
||||
|
||||
def _data_dir_has_content() -> bool:
|
||||
"""Whether anything predates this session in the plugin data directory."""
|
||||
try:
|
||||
for entry in memory_core.data_dir().iterdir():
|
||||
if entry.name != "install-state.json":
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _repair_install_state(path: Path) -> None:
|
||||
"""Rewrite an unparseable marker so version tracking can resume."""
|
||||
try:
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
temporary.write_text(
|
||||
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def claim_version_change() -> str | None:
|
||||
"""Return the previously recorded version if it differs, updating the marker.
|
||||
|
||||
Only meaningful once the marker exists — the first transition into 0.3.x has
|
||||
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
||||
the marker means the next session sees no change and records nothing.
|
||||
"""
|
||||
path = _install_state_path()
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
return None
|
||||
except json.JSONDecodeError:
|
||||
# A crash between O_EXCL and the write leaves an empty marker. Left
|
||||
# alone it disables every future upgrade event on this machine, because
|
||||
# claim_install sees the file and this function cannot parse it.
|
||||
state = None
|
||||
if not isinstance(state, dict):
|
||||
_repair_install_state(path)
|
||||
return None
|
||||
previous = str(state.get("plugin_version") or "")
|
||||
if not previous or previous == memory_core.PLUGIN_VERSION:
|
||||
return None
|
||||
# Claim the transition with an exclusive sentinel before rewriting the
|
||||
# marker. A plain read-modify-write let every concurrently starting session
|
||||
# observe the old version and each record its own upgrade — and the first
|
||||
# session after a version bump is exactly when several agent windows restart
|
||||
# together.
|
||||
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
||||
try:
|
||||
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
||||
state["upgraded_at"] = memory_core.utc_now()
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
try:
|
||||
temporary.write_text(json.dumps(state), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
# Release the claim. The marker still records the old version, so
|
||||
# without this the sentinel makes claim_version_change return early on
|
||||
# every later run and this version's upgrade is never recorded again.
|
||||
for leftover in (sentinel, temporary):
|
||||
try:
|
||||
leftover.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
return previous
|
||||
|
||||
|
||||
def record(
|
||||
@@ -168,19 +492,32 @@ def record(
|
||||
except OSError:
|
||||
pass
|
||||
properties = _safe_value(properties)
|
||||
# Stamped in the RECORDING process, beside harness. `source` used to be
|
||||
# read in the sending process from a module global, so whichever process
|
||||
# drained the spool named every event in it. flush() spreads per-event
|
||||
# properties last, so this now wins over any sender's default.
|
||||
properties.update(
|
||||
harness=_harness,
|
||||
source=_source_tag,
|
||||
plugin_version=memory_core.PLUGIN_VERSION,
|
||||
os=sys.platform,
|
||||
python_version=platform.python_version(),
|
||||
)
|
||||
# Assigned only when the digest is real. _scoped_digest returns "" when
|
||||
# the salt could not be persisted, and an empty property is worse than an
|
||||
# absent one: it survives the None filter below and reads as a value.
|
||||
if repo is not None:
|
||||
properties["repo_hash"] = _digest(getattr(repo, "identity", ""))
|
||||
repo_hash = _scoped_digest(getattr(repo, "identity", ""))
|
||||
if repo_hash:
|
||||
properties["repo_hash"] = repo_hash
|
||||
if session_id:
|
||||
properties["session_hash"] = _digest(session_id)
|
||||
session_hash = _scoped_digest(session_id)
|
||||
if session_hash:
|
||||
properties["session_hash"] = session_hash
|
||||
line = json.dumps(
|
||||
{
|
||||
"event": f"{EVENT_PREFIX}.{event}",
|
||||
"uuid": str(uuid.uuid4()),
|
||||
"timestamp": memory_core.utc_now(),
|
||||
"properties": {
|
||||
key: value for key, value in properties.items() if value is not None
|
||||
@@ -239,38 +576,201 @@ def spawn_flush() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _claim_name(attempt: int = 0) -> str:
|
||||
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
||||
only ever discards a batch that was actually retried and failed."""
|
||||
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
||||
|
||||
|
||||
def _claim_attempt(claim: Path) -> int:
|
||||
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape.
|
||||
|
||||
Anchored on field position, not on a leading "a": the legacy shape is
|
||||
``telemetry-<pid>-<hex>.sending`` and a hex id such as ``a1234567`` would
|
||||
otherwise parse as attempt 1234567 and be discarded unsent on the first
|
||||
flush after an upgrade.
|
||||
"""
|
||||
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
||||
parts = stem.split("-")
|
||||
if len(parts) != 4:
|
||||
return 0
|
||||
tail = parts[3]
|
||||
if tail.startswith("a") and tail[1:].isdigit():
|
||||
return int(tail[1:])
|
||||
return 0
|
||||
|
||||
|
||||
def _touch(path: Path) -> None:
|
||||
"""Refresh mtime so a claim's age measures time since it was claimed.
|
||||
|
||||
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
||||
after a quiet minute inherited the spool's last-write time and looked
|
||||
abandoned the instant it was made. A second sender would then take it over
|
||||
while the first was still posting, and both would deliver the batch.
|
||||
"""
|
||||
try:
|
||||
os.utime(path, None)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _claim_spool() -> Path | None:
|
||||
"""Rename the spool aside so exactly one sender owns each batch."""
|
||||
directory = memory_core.data_dir()
|
||||
claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending"
|
||||
claim = directory / _claim_name()
|
||||
spool = _spool_path()
|
||||
try:
|
||||
spool.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
pass
|
||||
return _claim_parked(directory)
|
||||
|
||||
|
||||
def _sweep_debris(directory: Path) -> None:
|
||||
"""Remove files nothing else will ever pick up again.
|
||||
|
||||
*.partial is a temp file orphaned by a crash between write and rename.
|
||||
*.corrupt is a batch quarantined for undecodable content. No glob in this
|
||||
module matches either, so without this they accumulate on disk for the life
|
||||
of the install.
|
||||
|
||||
Quarantined batches are kept far longer than debris: they are the only
|
||||
evidence left of events that could not be delivered, and someone diagnosing
|
||||
a report of missing telemetry has to be able to find one.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending")):
|
||||
for debris in directory.glob("telemetry-*.partial"):
|
||||
try:
|
||||
if now - debris.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
debris.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
for quarantined in directory.glob("telemetry-*.corrupt"):
|
||||
try:
|
||||
if now - quarantined.stat().st_mtime > CLAIM_EXPIRY_SECONDS:
|
||||
quarantined.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
# The same reasoning covers *.tmp. _write_identity and _install_salt both
|
||||
# create one and unlink it in a finally, which a SIGKILL skips, and no glob
|
||||
# in this module matches the leftovers either.
|
||||
for temporary in directory.glob("telemetry-*.tmp"):
|
||||
try:
|
||||
if now - temporary.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
|
||||
def _claim_parked(directory: Path) -> Path | None:
|
||||
"""Take the oldest abandoned claim, if any lease has actually expired.
|
||||
|
||||
Kept separate from the live spool so flush() can drain both in one run.
|
||||
Previously parked batches were only reachable when no spool existed at all,
|
||||
and because sessions keep recording there usually was one — so a batch
|
||||
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
||||
even though its own presence is what started the sender.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
||||
try:
|
||||
age = now - orphan.stat().st_mtime
|
||||
except OSError:
|
||||
continue
|
||||
if age > CLAIM_EXPIRY_SECONDS:
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
# Someone else holds a live lease on it. This check has to come
|
||||
# first. Claiming a file bumps its attempt count and refreshes its
|
||||
# mtime, so a sender that has just taken the final attempt looks
|
||||
# exhausted to everyone else while it is actively draining. Judging
|
||||
# exhaustion before liveness let a second sender unlink a batch out
|
||||
# from under its owner, losing every event in it.
|
||||
continue
|
||||
# Attempts, not age. Every re-claim touches the mtime and every release
|
||||
# backdates it by a fixed amount, so age is pinned near the stale
|
||||
# threshold and never reaches the expiry. Age stays only as a backstop
|
||||
# for files that never carried an attempt marker.
|
||||
if _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS or age > CLAIM_EXPIRY_SECONDS:
|
||||
try:
|
||||
orphan.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
continue
|
||||
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
||||
try:
|
||||
orphan.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _safe_mtime(path: Path) -> float:
|
||||
try:
|
||||
return path.stat().st_mtime
|
||||
except OSError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
||||
"""Persist the unsent remainder, atomically, and refresh the lease.
|
||||
|
||||
Called after every successful batch. Two jobs: a retry resumes where the
|
||||
send stopped instead of re-posting from the top, and the rewrite doubles as
|
||||
the lease heartbeat, so a slow sender does not have its claim stolen
|
||||
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
||||
"""
|
||||
if not remaining:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return True
|
||||
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
||||
try:
|
||||
payload = "".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining)
|
||||
# fsync before the rename: without it the rename can land while the
|
||||
# bytes have not, and the claim comes back empty or truncated after a
|
||||
# crash. _drain then reads zero events and unlinks it.
|
||||
with open(temporary, "w", encoding="utf-8") as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
temporary.replace(claim)
|
||||
_touch(claim)
|
||||
return True
|
||||
except OSError:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
||||
"""Persist the remainder and drop the lease, because this sender has given up.
|
||||
|
||||
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
||||
make an abandoned batch look actively owned for a further
|
||||
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
||||
threshold lets the next flush pick it up immediately, while the attempt
|
||||
count in the filename still bounds how many times that can happen.
|
||||
"""
|
||||
if not _rewrite_claim(claim, remaining):
|
||||
return
|
||||
try:
|
||||
# Backdate past the stale threshold so the next flush can pick it up,
|
||||
# minus a cooldown that grows with the attempts already spent. Clamped so
|
||||
# the mtime never lands in the future, which would read as a live lease.
|
||||
cooldown = min(_claim_attempt(claim) * RETRY_COOLDOWN_SECONDS, CLAIM_STALE_SECONDS)
|
||||
released = time.time() - CLAIM_STALE_SECONDS - 1 + cooldown
|
||||
os.utime(claim, (released, released))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_email(key: str) -> str:
|
||||
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
||||
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
||||
@@ -300,34 +800,130 @@ def _post(payload: dict[str, Any], url: str) -> bool:
|
||||
|
||||
|
||||
def resolve_distinct_id() -> tuple[str, str]:
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any."""
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
||||
|
||||
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
||||
anonymous id: aliasing one account email to another merges two real person
|
||||
profiles and cannot be undone, so a key that now belongs to a different
|
||||
account re-resolves with no alias.
|
||||
"""
|
||||
identity = _read_identity()
|
||||
email = identity.get("email", "")
|
||||
if email:
|
||||
return email, ""
|
||||
key = memory_core.api_key()
|
||||
fingerprint = _digest(key) if key else ""
|
||||
email = identity.get("email", "")
|
||||
|
||||
if email and fingerprint:
|
||||
recorded = identity.get("key_fingerprint", "")
|
||||
if recorded == fingerprint:
|
||||
return email, ""
|
||||
if not recorded:
|
||||
# Rows written before fingerprints existed. Verify rather than
|
||||
# adopt: a key changed before the upgrade would otherwise bind the
|
||||
# new key to the previous account's email, permanently, and the
|
||||
# fingerprint would then agree with itself forever after.
|
||||
verified = _resolve_email(key)
|
||||
if not verified:
|
||||
# Offline, firewalled, or the API is down. Keep the previous
|
||||
# behaviour and retry on the next flush rather than dropping a
|
||||
# real account attribution. Safe because the same network that
|
||||
# failed /v1/ping/ is about to fail the PostHog POST, so nothing
|
||||
# is delivered under the unverified identity in the meantime.
|
||||
return email, ""
|
||||
identity["email"] = verified
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return verified, ""
|
||||
|
||||
if not key:
|
||||
# No key to verify the account with; do not keep attributing to it.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
email = _resolve_email(key)
|
||||
if not email:
|
||||
|
||||
resolved = _resolve_email(key)
|
||||
if not resolved:
|
||||
# The key changed and will not resolve (revoked, offline, API down).
|
||||
# Reaching here with an email means the recorded fingerprint disagreed,
|
||||
# so the key really did change. Drop the account and rotate: the stored
|
||||
# anonymous id may already be merged into that account's person, and
|
||||
# reusing it would keep the events on the profile we are trying to
|
||||
# leave.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
previous = identity.get("anonymous_id", "")
|
||||
identity["email"] = email
|
||||
|
||||
# Alias only when going anonymous -> email for the first time. Once an anon
|
||||
# id has been merged into an account it must never be offered again: an
|
||||
# alias naming an already-identified id is what could link two real people.
|
||||
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
||||
if previous:
|
||||
identity["aliased"] = True
|
||||
identity["email"] = resolved
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return email, previous
|
||||
return resolved, previous
|
||||
|
||||
|
||||
def flush() -> int:
|
||||
"""Drain claimed spools to PostHog and return the number of events sent."""
|
||||
"""Drain the live spool, then any parked claims, and return events sent."""
|
||||
if not is_enabled():
|
||||
return 0
|
||||
claim = _claim_spool()
|
||||
sent, delivered = _drain(_claim_spool())
|
||||
if not delivered:
|
||||
# The network is failing. Retrying other batches now would only burn
|
||||
# their attempt budget against the same broken connection.
|
||||
return sent
|
||||
|
||||
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
||||
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
||||
directory = memory_core.data_dir()
|
||||
_sweep_debris(directory)
|
||||
for _ in range(MAX_PARKED_PER_RUN):
|
||||
parked = _claim_parked(directory)
|
||||
if parked is None:
|
||||
break
|
||||
count, delivered = _drain(parked)
|
||||
sent += count
|
||||
if not delivered:
|
||||
break
|
||||
return sent
|
||||
|
||||
|
||||
def _drain(claim: Path | None) -> tuple[int, bool]:
|
||||
"""Post one claimed batch file, recording progress after every batch.
|
||||
|
||||
Returns (events sent, whether everything was delivered).
|
||||
"""
|
||||
if claim is None:
|
||||
return 0
|
||||
return 0, True
|
||||
try:
|
||||
lines = claim.read_text(encoding="utf-8").splitlines()
|
||||
except ValueError:
|
||||
# UnicodeDecodeError from a torn write: the content is unrecoverable, so
|
||||
# quarantine rather than retry. flush() runs from a bare `finally:` in
|
||||
# flush_worker, so raising here also skips the handoff cleanup, and an
|
||||
# undecodable file would otherwise be re-read on every flush forever.
|
||||
# Reported as delivered because there is nothing left to deliver and the
|
||||
# rest of the run should continue.
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt"))
|
||||
except OSError:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0, True
|
||||
except OSError:
|
||||
return 0
|
||||
# Could not read it, which is not the same as having nothing to send.
|
||||
# The file is left exactly where it is: a vanished or briefly unreadable
|
||||
# claim is retryable, and quarantining it here would discard events over
|
||||
# a transient filesystem error. Reported as undelivered so the run stops
|
||||
# instead of counting a batch nothing was posted from as delivered.
|
||||
return 0, False
|
||||
events = []
|
||||
for line in lines:
|
||||
try:
|
||||
@@ -337,11 +933,18 @@ def flush() -> int:
|
||||
if isinstance(value, dict) and value.get("event"):
|
||||
events.append(value)
|
||||
if not events:
|
||||
# Only delete when the file really is empty. A non-empty file that
|
||||
# parses to nothing is a torn write, and its contents are the unsent
|
||||
# remainder — deleting it is the data loss this PR exists to prevent.
|
||||
try:
|
||||
claim.unlink()
|
||||
empty = claim.stat().st_size == 0
|
||||
except OSError:
|
||||
empty = True
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt")) if not empty else claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
return 0, True
|
||||
|
||||
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
||||
if aliased_anonymous_id:
|
||||
@@ -360,12 +963,17 @@ def flush() -> int:
|
||||
|
||||
sent = 0
|
||||
for start in range(0, len(events), BATCH_SIZE):
|
||||
chunk = events[start : start + BATCH_SIZE]
|
||||
batch = [
|
||||
{
|
||||
"event": event["event"],
|
||||
"distinct_id": distinct_id,
|
||||
# Carried through from record() so a resend can be collapsed.
|
||||
"uuid": event.get("uuid"),
|
||||
"timestamp": event.get("timestamp"),
|
||||
"properties": {
|
||||
# Fallback only: events recorded by a build before source
|
||||
# moved into record() have none of their own.
|
||||
"source": _source_tag,
|
||||
"language": "python",
|
||||
"$process_person_profile": False,
|
||||
@@ -373,16 +981,24 @@ def flush() -> int:
|
||||
**(event.get("properties") or {}),
|
||||
},
|
||||
}
|
||||
for event in events[start : start + BATCH_SIZE]
|
||||
for event in chunk
|
||||
]
|
||||
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
||||
return sent
|
||||
sent += len(batch)
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return sent
|
||||
# Keep only what has not been delivered, and release the lease.
|
||||
# Previously the whole file was kept and the retry re-posted every
|
||||
# batch, including the ones that had already arrived.
|
||||
_release_claim(claim, events[start:])
|
||||
return sent, False
|
||||
sent += len(chunk)
|
||||
# Record progress and refresh the lease after each successful batch, so
|
||||
# a crash repeats at most one batch instead of the entire file. If the
|
||||
# rewrite fails the claim still holds delivered events, so stop rather
|
||||
# than carry on as though progress were recorded — continuing is how the
|
||||
# duplicate delivery this PR fixes would come back.
|
||||
if not _rewrite_claim(claim, events[start + len(chunk) :]):
|
||||
_release_claim(claim, events[start + len(chunk) :])
|
||||
return sent, False
|
||||
return sent, True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"id": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"homepage": "https://docs.mem0.ai/integrations/antigravity",
|
||||
"native": {
|
||||
"pluginRoot": "${ANTIGRAVITY_PLUGIN_ROOT}",
|
||||
|
||||
@@ -7,8 +7,8 @@ disable-model-invocation: true
|
||||
# Pause memory capture
|
||||
|
||||
To pause (hooks stop capturing and sending session content; a minimal
|
||||
anonymous telemetry ping still fires at session start unless
|
||||
`MEM0_TELEMETRY=false`):
|
||||
telemetry ping still fires at session start, under your Mem0 account email,
|
||||
unless `MEM0_TELEMETRY=false`):
|
||||
|
||||
```bash
|
||||
python3 "${ANTIGRAVITY_PLUGIN_ROOT}/core/memory_cli.py" --harness "antigravity" pause
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"author": {
|
||||
"name": "Mem0"
|
||||
|
||||
@@ -136,13 +136,23 @@ Local data lives in `${CLAUDE_PLUGIN_DATA}`:
|
||||
- `pending/`: sessions waiting to be sent to Mem0 (retried after interruption)
|
||||
- `flush-worker.log`: whether memory creation succeeded
|
||||
- `plugin-errors.log`: hook errors (no credentials)
|
||||
- `telemetry.jsonl` / `telemetry-identity.json`: anonymous usage events
|
||||
- `telemetry.jsonl` / `telemetry-identity.json`: usage events and the id they are sent under
|
||||
- `telemetry-salt`: random per-install salt for the repo and session hashes
|
||||
- `install-state.json`: records that install has been counted once on this machine
|
||||
|
||||
Mem0 receives captured user messages, Claude's answers, sidekick assignments and completed responses, and changed file paths. When a failed command is recorded, extraction can also include bounded command details and results. Complete files and general tool output stay on your machine. Values that look like credentials are redacted before anything is sent.
|
||||
|
||||
## Telemetry
|
||||
|
||||
Anonymous usage events (which hook ran, timing, result counts, failure types) so Mem0 can identify what's used and what's breaking. Repo and session IDs are hashed before leaving your machine. Prompts, memory text, file paths, tool output, and API keys are never sent.
|
||||
Usage events (which hook ran, timing, result counts, failure types) so Mem0 can identify what's used and what's breaking.
|
||||
|
||||
**These events are not anonymous.** When an API key is configured — which installing the plugin requires — events are sent under your Mem0 account email, the same way the Python SDK and the CLI attribute theirs. Without a key they are sent under a random per-machine id.
|
||||
|
||||
What each event carries: the event name, the plugin version, the harness it ran in, your OS and Python version, and per-event properties describing what happened — timings, counts, coarse outcome and failure labels, and which model was configured. Repository and session identifiers are hashed with a random salt generated on your machine, so they cannot be linked back to a repository name or path.
|
||||
|
||||
Rather than restate a list that drifts, the exact set is enforced in code: `telemetry.record` filters every property through a denylist of sensitive keys and redacts credential-shaped values. See `_PRIVATE_KEYS` in `core/telemetry.py`.
|
||||
|
||||
Prompts, memory text, queries, file paths, repository names, and API keys are never sent.
|
||||
|
||||
Turn it off:
|
||||
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Generated by integrations/agent-plugin-core/build/build.py. Do not edit."""
|
||||
|
||||
HARNESS_ID = "claude-code"
|
||||
SOURCE_TAG = "CLAUDE_CODE_PLUGIN"
|
||||
|
||||
# Platform-side vocabulary (mem0_event.source + X-Application). The whole
|
||||
# plugin family is one source; which editor it runs in is the application.
|
||||
# An empty application means the host is unknown, and memory_core omits
|
||||
# the header entirely rather than sending a placeholder.
|
||||
PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
PLATFORM_APPLICATION = "claude-code"
|
||||
@@ -290,6 +290,11 @@ def run(
|
||||
if args.plugin_data_dir:
|
||||
os.environ[data_dir_env] = args.plugin_data_dir
|
||||
|
||||
# Snapshot BEFORE anything writes to the data dir: cache_plugin_api_key
|
||||
# writes `api-key` and EvidenceStore creates `evidence.sqlite3`, so asking
|
||||
# after them always saw content and every fresh install reported an upgrade.
|
||||
data_dir_was_empty = telemetry.data_dir_was_empty()
|
||||
|
||||
cache_plugin_api_key()
|
||||
if args.action == "session-start":
|
||||
clear_stale_api_key_cache()
|
||||
@@ -305,8 +310,19 @@ def run(
|
||||
return 0
|
||||
|
||||
if args.action == "session-start":
|
||||
if telemetry.is_first_run():
|
||||
# Claims the marker atomically and says which event to record, so a
|
||||
# second session starting alongside this one cannot record it too.
|
||||
first_event = telemetry.claim_install(was_empty=data_dir_was_empty)
|
||||
if first_event == "install":
|
||||
telemetry.record("install")
|
||||
elif first_event == "upgrade":
|
||||
# First run after a build that never wrote the marker; the
|
||||
# predecessor version was never recorded anywhere.
|
||||
telemetry.record("upgrade", from_version="pre-0.3")
|
||||
else:
|
||||
previous = telemetry.claim_version_change()
|
||||
if previous:
|
||||
telemetry.record("upgrade", from_version=previous)
|
||||
recovered = recover_pending_handoffs()
|
||||
record_session_start(store, hook_input)
|
||||
if recovered:
|
||||
|
||||
@@ -29,7 +29,7 @@ from typing import Any, Iterable
|
||||
import telemetry
|
||||
|
||||
DEFAULT_API_URL = "https://api.mem0.ai"
|
||||
PLUGIN_VERSION = "0.3.1"
|
||||
PLUGIN_VERSION = "0.3.2"
|
||||
|
||||
_harness_name: str = "generic"
|
||||
_harness_env_prefix: str = "MEM0_PLUGIN"
|
||||
@@ -1800,6 +1800,34 @@ def extraction_message_batches(
|
||||
return batches
|
||||
|
||||
|
||||
# Platform surface attribution. Read from the generated per-host module so a new
|
||||
# entrypoint is correct without remembering to configure anything.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
except ImportError:
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
|
||||
def platform_headers(key: str) -> dict[str, str]:
|
||||
"""Auth plus the three surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are set-once by contract: this is the
|
||||
outermost layer, so it sets them, and nothing below may overwrite them.
|
||||
X-Mem0-Client is append-only — anything downstream adds itself to the tail.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": _PLATFORM_SOURCE,
|
||||
"X-Mem0-Client": f"mem0-plugin/{PLUGIN_VERSION}",
|
||||
}
|
||||
if _PLATFORM_APPLICATION:
|
||||
headers["X-Application"] = _PLATFORM_APPLICATION
|
||||
return headers
|
||||
|
||||
|
||||
def _request_json(
|
||||
url: str, key: str, payload: dict[str, Any], timeout: float
|
||||
) -> tuple[dict[str, Any] | list[Any], int, int]:
|
||||
@@ -1807,7 +1835,7 @@ def _request_json(
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1834,7 +1862,7 @@ def _get_json(
|
||||
) -> tuple[dict[str, Any] | list[Any], int]:
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="GET",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1980,6 +2008,13 @@ def flush_session(
|
||||
"user_id": write_user,
|
||||
"app_id": repo.app_id,
|
||||
"run_id": session_id,
|
||||
# Top level, not metadata: the backend reads `source` from the body or
|
||||
# the query string, never from metadata, which is where this used to
|
||||
# sit. The X-Mem0-Source header is also read, but only from the
|
||||
# platform release that ships alongside this change, so the body value
|
||||
# is what makes attribution work on both. The harness tag stays in
|
||||
# metadata as hook provenance.
|
||||
"source": _PLATFORM_SOURCE,
|
||||
"metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)},
|
||||
"agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS,
|
||||
"custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS,
|
||||
@@ -2523,7 +2558,7 @@ def _collect_memory_ids(
|
||||
def _delete_memory(api_url: str, key: str, memory_id: str) -> bool:
|
||||
request = urllib.request.Request(
|
||||
f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/",
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Anonymous usage telemetry for Mem0 agent plugins.
|
||||
"""Usage telemetry for Mem0 agent plugins.
|
||||
|
||||
Events are linked to your Mem0 account email when an API key is configured, and
|
||||
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
||||
attribute the same way.
|
||||
|
||||
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
||||
touches the network: `record` appends one JSON line to a local spool and returns.
|
||||
@@ -9,7 +13,8 @@ started once per session and again from the flush worker that is already detache
|
||||
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
||||
|
||||
Never sends prompts, memory text, queries, file paths, repository names, or API
|
||||
keys: only event names, durations, counts, coarse outcomes, and salted hashes.
|
||||
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
||||
identifiers hashed with a random per-install salt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,8 +34,24 @@ from typing import Any
|
||||
|
||||
import memory_core
|
||||
|
||||
_harness: str = "generic"
|
||||
_source_tag: str = "MEM0_PLUGIN"
|
||||
# Seeded from the per-host module the build generates into core/. Two processes
|
||||
# in this pipeline never call init() — mcp_server.py, and the detached
|
||||
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
||||
# was what every one of their events got labelled with.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
||||
except ImportError:
|
||||
_DEFAULT_HARNESS = "generic"
|
||||
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
_salt_cache: str = ""
|
||||
_harness: str = _DEFAULT_HARNESS
|
||||
_source_tag: str = _DEFAULT_SOURCE_TAG
|
||||
_PRIVATE_KEYS = {
|
||||
"apikey",
|
||||
"authorization",
|
||||
@@ -56,10 +77,19 @@ _PRIVATE_KEYS = {
|
||||
}
|
||||
|
||||
|
||||
def init(harness: str = "generic", source_tag: str = "") -> None:
|
||||
def init(harness: str = "", source_tag: str = "") -> None:
|
||||
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
||||
|
||||
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
||||
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
||||
meant one plugin could emit three different source values depending on which
|
||||
process happened to send the batch.
|
||||
"""
|
||||
global _harness, _source_tag
|
||||
_harness = harness
|
||||
_source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN"
|
||||
_harness = harness or _DEFAULT_HARNESS
|
||||
_source_tag = source_tag or (
|
||||
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
||||
)
|
||||
|
||||
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
||||
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
||||
@@ -70,6 +100,16 @@ BATCH_SIZE = 100
|
||||
SEND_TIMEOUT = 5
|
||||
CLAIM_STALE_SECONDS = 120
|
||||
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
||||
# A batch is only discarded once it has genuinely been retried this many times.
|
||||
MAX_CLAIM_ATTEMPTS = 3
|
||||
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
||||
# cannot turn one flush into an unbounded send loop.
|
||||
MAX_PARKED_PER_RUN = 3
|
||||
# Added to the wait before a released claim becomes reclaimable, per attempt
|
||||
# already spent. Releasing straight to "reclaimable now" let two senders burn the
|
||||
# whole budget within seconds of one another on a single momentary failure, and
|
||||
# discard a batch a retry a minute later would have delivered.
|
||||
RETRY_COOLDOWN_SECONDS = 60
|
||||
|
||||
|
||||
def is_enabled() -> bool:
|
||||
@@ -83,9 +123,126 @@ def is_enabled() -> bool:
|
||||
|
||||
|
||||
def _digest(value: str, length: int = 16) -> str:
|
||||
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _salt_path() -> Path:
|
||||
return memory_core.data_dir() / "telemetry-salt"
|
||||
|
||||
|
||||
def _install_salt() -> str:
|
||||
"""Random per-install salt, created once and memoized for the process.
|
||||
|
||||
Deliberately its own file, claimed with O_CREAT|O_EXCL, rather than a key in
|
||||
the identity file. Three reasons, all of which produced wrong data when this
|
||||
lived in the identity dict:
|
||||
|
||||
- Hooks are short-lived separate processes firing on every tool call, and
|
||||
people run more than one agent window. A read-modify-write would let each
|
||||
process mint its own salt, so one repository would hash several ways in the
|
||||
window before a writer won.
|
||||
- resolve_distinct_id holds a copy of the identity dict across a network call
|
||||
to /v1/ping/, so whichever write landed second erased the other's key —
|
||||
losing either the salt (repo_hash changes mid-stream) or the email (a
|
||||
second $identify, splitting the person).
|
||||
- Touching the identity file from record() would create it, and is_first_run
|
||||
keys off that file, so recording an event would silently suppress the
|
||||
install event.
|
||||
|
||||
Published atomically, and there is deliberately no derived fallback. Creating
|
||||
the file with O_CREAT|O_EXCL and then writing into it leaves a window where
|
||||
the file exists and is empty, and a concurrent hook that reads it in that
|
||||
window gets nothing. Falling back to a digest of the path would hand that
|
||||
process a salt an attacker can compute, memoized for its whole run, which is
|
||||
the privacy control this function exists to provide silently turning itself
|
||||
off under load. The salt is written to a private temp file first and linked
|
||||
into place, so the name either does not exist or already has the full value.
|
||||
|
||||
Returns "" when it genuinely cannot persist. Callers omit the hash entirely
|
||||
rather than emit an unsalted one.
|
||||
"""
|
||||
global _salt_cache
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
|
||||
path = _salt_path()
|
||||
# Read before writing. Hooks are separate processes firing on every tool
|
||||
# call, so all but the first find the salt already published; going straight
|
||||
# to create-fsync-link-unlink meant every one of them paid an fsync to
|
||||
# discover that, on a path whose whole promise is appending a line and
|
||||
# returning.
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
temporary = path.with_name(f"{path.name}.{os.getpid()}.tmp")
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
stream.write(uuid.uuid4().hex)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
try:
|
||||
# Atomic claim: fails if another process already published one.
|
||||
# os.link rather than replace, which would clobber theirs.
|
||||
os.link(temporary, path)
|
||||
except FileExistsError:
|
||||
pass
|
||||
except OSError:
|
||||
# No hardlinks here (some network mounts, some container volumes).
|
||||
# Claim the name directly instead. That reopens the empty-file
|
||||
# window, but the window is now benign: a reader that lands in it
|
||||
# gets "" and omits the hash for that process rather than caching a
|
||||
# guessable one. Losing the hashes on every run of an entire
|
||||
# filesystem is the worse failure.
|
||||
try:
|
||||
fallback = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fallback, "w", encoding="utf-8") as stream:
|
||||
stream.write(temporary.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_salt_cache = ""
|
||||
return _salt_cache
|
||||
|
||||
|
||||
def _scoped_digest(value: str, length: int = 16) -> str:
|
||||
"""Salted digest for values drawn from a guessable space.
|
||||
|
||||
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
||||
no remote — which normally contains the account username. Sixteen unsalted
|
||||
hex characters over that input space is enumerable, so this is not a
|
||||
privacy control without the salt. Salting per install keeps every
|
||||
within-account join the analytics actually use and gives up only
|
||||
cross-machine joins on the same repository, which nothing computes.
|
||||
|
||||
Returns "" when there is no salt, so record() omits the property. An
|
||||
unsalted digest over this input space is close to plaintext, and emitting one
|
||||
under a name that implies it is hashed is worse than sending nothing.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
salt = _install_salt()
|
||||
if not salt:
|
||||
return ""
|
||||
return hashlib.sha256(f"{salt}:{value}".encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _safe_value(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return memory_core.redact(value)
|
||||
@@ -145,9 +302,176 @@ def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
||||
return created
|
||||
|
||||
|
||||
def _rotate_anonymous_id(identity: dict[str, str]) -> str:
|
||||
"""Mint a fresh anonymous id because the account context is gone.
|
||||
|
||||
The previous id may already have been merged into a person profile by an
|
||||
$identify, and that merge is permanent. Reusing it after a logout or a key
|
||||
change attributes everything that follows to the account that just went
|
||||
away, which is the same misattribution the key fingerprint exists to stop,
|
||||
only arriving through the anonymous path instead.
|
||||
|
||||
`aliased` is cleared with it: the new id has never been merged, so it is
|
||||
eligible to be aliased into whatever account comes next.
|
||||
"""
|
||||
created = f"code-anon-{uuid.uuid4().hex}"
|
||||
identity["anonymous_id"] = created
|
||||
identity.pop("aliased", None)
|
||||
_write_identity(identity)
|
||||
return created
|
||||
|
||||
|
||||
def _install_state_path() -> Path:
|
||||
return memory_core.data_dir() / "install-state.json"
|
||||
|
||||
|
||||
def is_first_run() -> bool:
|
||||
"""Whether this machine has never recorded a plugin event before."""
|
||||
return not _identity_path().exists()
|
||||
"""Whether install has never been recorded on this machine.
|
||||
|
||||
Deliberately NOT the identity file. That file is only written by a
|
||||
successful flush, so an offline or firewalled user recorded code.install on
|
||||
every single session, forever — and every 0.2.x user recorded one on their
|
||||
first 0.3.x session because 0.2.x never wrote it at all.
|
||||
"""
|
||||
return not _install_state_path().exists()
|
||||
|
||||
|
||||
def data_dir_was_empty() -> bool:
|
||||
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
||||
|
||||
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
||||
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
||||
asking at claim time always saw content and every fresh install reported an
|
||||
upgrade. The caller snapshots this at the top of the run instead.
|
||||
"""
|
||||
return not _data_dir_has_content()
|
||||
|
||||
|
||||
def claim_install(was_empty: bool | None = None) -> str | None:
|
||||
"""Claim the one install/upgrade record for this machine, atomically.
|
||||
|
||||
Returns the event to record ("install" or "upgrade"), or None if another
|
||||
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
||||
cannot both win.
|
||||
|
||||
`was_empty` must come from data_dir_was_empty() called before this process
|
||||
wrote anything. Omitting it falls back to checking now, which is only
|
||||
correct for a caller that has touched nothing.
|
||||
"""
|
||||
if not is_enabled():
|
||||
# Never consume the one-shot claim while the user is opted out, or they
|
||||
# would silently lose their install event if they later opt in.
|
||||
return None
|
||||
|
||||
path = _install_state_path()
|
||||
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
json.dump(
|
||||
{
|
||||
"plugin_version": memory_core.PLUGIN_VERSION,
|
||||
"installed_at": memory_core.utc_now(),
|
||||
"upgraded": upgrading,
|
||||
},
|
||||
stream,
|
||||
)
|
||||
# Durable before this returns. The O_EXCL open is what makes the
|
||||
# claim exclusive, so it cannot be replaced by a temp-and-rename
|
||||
# without losing that, which leaves the content as the thing to make
|
||||
# safe. A kill between the open and this fsync used to leave a marker
|
||||
# that exists but parses to nothing: is_first_run reads it as claimed
|
||||
# and claim_version_change cannot read a version out of it.
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
except OSError:
|
||||
pass
|
||||
return "upgrade" if upgrading else "install"
|
||||
|
||||
|
||||
def _data_dir_has_content() -> bool:
|
||||
"""Whether anything predates this session in the plugin data directory."""
|
||||
try:
|
||||
for entry in memory_core.data_dir().iterdir():
|
||||
if entry.name != "install-state.json":
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _repair_install_state(path: Path) -> None:
|
||||
"""Rewrite an unparseable marker so version tracking can resume."""
|
||||
try:
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
temporary.write_text(
|
||||
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def claim_version_change() -> str | None:
|
||||
"""Return the previously recorded version if it differs, updating the marker.
|
||||
|
||||
Only meaningful once the marker exists — the first transition into 0.3.x has
|
||||
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
||||
the marker means the next session sees no change and records nothing.
|
||||
"""
|
||||
path = _install_state_path()
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
return None
|
||||
except json.JSONDecodeError:
|
||||
# A crash between O_EXCL and the write leaves an empty marker. Left
|
||||
# alone it disables every future upgrade event on this machine, because
|
||||
# claim_install sees the file and this function cannot parse it.
|
||||
state = None
|
||||
if not isinstance(state, dict):
|
||||
_repair_install_state(path)
|
||||
return None
|
||||
previous = str(state.get("plugin_version") or "")
|
||||
if not previous or previous == memory_core.PLUGIN_VERSION:
|
||||
return None
|
||||
# Claim the transition with an exclusive sentinel before rewriting the
|
||||
# marker. A plain read-modify-write let every concurrently starting session
|
||||
# observe the old version and each record its own upgrade — and the first
|
||||
# session after a version bump is exactly when several agent windows restart
|
||||
# together.
|
||||
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
||||
try:
|
||||
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
||||
state["upgraded_at"] = memory_core.utc_now()
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
try:
|
||||
temporary.write_text(json.dumps(state), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
# Release the claim. The marker still records the old version, so
|
||||
# without this the sentinel makes claim_version_change return early on
|
||||
# every later run and this version's upgrade is never recorded again.
|
||||
for leftover in (sentinel, temporary):
|
||||
try:
|
||||
leftover.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
return previous
|
||||
|
||||
|
||||
def record(
|
||||
@@ -168,19 +492,32 @@ def record(
|
||||
except OSError:
|
||||
pass
|
||||
properties = _safe_value(properties)
|
||||
# Stamped in the RECORDING process, beside harness. `source` used to be
|
||||
# read in the sending process from a module global, so whichever process
|
||||
# drained the spool named every event in it. flush() spreads per-event
|
||||
# properties last, so this now wins over any sender's default.
|
||||
properties.update(
|
||||
harness=_harness,
|
||||
source=_source_tag,
|
||||
plugin_version=memory_core.PLUGIN_VERSION,
|
||||
os=sys.platform,
|
||||
python_version=platform.python_version(),
|
||||
)
|
||||
# Assigned only when the digest is real. _scoped_digest returns "" when
|
||||
# the salt could not be persisted, and an empty property is worse than an
|
||||
# absent one: it survives the None filter below and reads as a value.
|
||||
if repo is not None:
|
||||
properties["repo_hash"] = _digest(getattr(repo, "identity", ""))
|
||||
repo_hash = _scoped_digest(getattr(repo, "identity", ""))
|
||||
if repo_hash:
|
||||
properties["repo_hash"] = repo_hash
|
||||
if session_id:
|
||||
properties["session_hash"] = _digest(session_id)
|
||||
session_hash = _scoped_digest(session_id)
|
||||
if session_hash:
|
||||
properties["session_hash"] = session_hash
|
||||
line = json.dumps(
|
||||
{
|
||||
"event": f"{EVENT_PREFIX}.{event}",
|
||||
"uuid": str(uuid.uuid4()),
|
||||
"timestamp": memory_core.utc_now(),
|
||||
"properties": {
|
||||
key: value for key, value in properties.items() if value is not None
|
||||
@@ -239,38 +576,201 @@ def spawn_flush() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _claim_name(attempt: int = 0) -> str:
|
||||
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
||||
only ever discards a batch that was actually retried and failed."""
|
||||
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
||||
|
||||
|
||||
def _claim_attempt(claim: Path) -> int:
|
||||
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape.
|
||||
|
||||
Anchored on field position, not on a leading "a": the legacy shape is
|
||||
``telemetry-<pid>-<hex>.sending`` and a hex id such as ``a1234567`` would
|
||||
otherwise parse as attempt 1234567 and be discarded unsent on the first
|
||||
flush after an upgrade.
|
||||
"""
|
||||
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
||||
parts = stem.split("-")
|
||||
if len(parts) != 4:
|
||||
return 0
|
||||
tail = parts[3]
|
||||
if tail.startswith("a") and tail[1:].isdigit():
|
||||
return int(tail[1:])
|
||||
return 0
|
||||
|
||||
|
||||
def _touch(path: Path) -> None:
|
||||
"""Refresh mtime so a claim's age measures time since it was claimed.
|
||||
|
||||
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
||||
after a quiet minute inherited the spool's last-write time and looked
|
||||
abandoned the instant it was made. A second sender would then take it over
|
||||
while the first was still posting, and both would deliver the batch.
|
||||
"""
|
||||
try:
|
||||
os.utime(path, None)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _claim_spool() -> Path | None:
|
||||
"""Rename the spool aside so exactly one sender owns each batch."""
|
||||
directory = memory_core.data_dir()
|
||||
claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending"
|
||||
claim = directory / _claim_name()
|
||||
spool = _spool_path()
|
||||
try:
|
||||
spool.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
pass
|
||||
return _claim_parked(directory)
|
||||
|
||||
|
||||
def _sweep_debris(directory: Path) -> None:
|
||||
"""Remove files nothing else will ever pick up again.
|
||||
|
||||
*.partial is a temp file orphaned by a crash between write and rename.
|
||||
*.corrupt is a batch quarantined for undecodable content. No glob in this
|
||||
module matches either, so without this they accumulate on disk for the life
|
||||
of the install.
|
||||
|
||||
Quarantined batches are kept far longer than debris: they are the only
|
||||
evidence left of events that could not be delivered, and someone diagnosing
|
||||
a report of missing telemetry has to be able to find one.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending")):
|
||||
for debris in directory.glob("telemetry-*.partial"):
|
||||
try:
|
||||
if now - debris.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
debris.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
for quarantined in directory.glob("telemetry-*.corrupt"):
|
||||
try:
|
||||
if now - quarantined.stat().st_mtime > CLAIM_EXPIRY_SECONDS:
|
||||
quarantined.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
# The same reasoning covers *.tmp. _write_identity and _install_salt both
|
||||
# create one and unlink it in a finally, which a SIGKILL skips, and no glob
|
||||
# in this module matches the leftovers either.
|
||||
for temporary in directory.glob("telemetry-*.tmp"):
|
||||
try:
|
||||
if now - temporary.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
|
||||
def _claim_parked(directory: Path) -> Path | None:
|
||||
"""Take the oldest abandoned claim, if any lease has actually expired.
|
||||
|
||||
Kept separate from the live spool so flush() can drain both in one run.
|
||||
Previously parked batches were only reachable when no spool existed at all,
|
||||
and because sessions keep recording there usually was one — so a batch
|
||||
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
||||
even though its own presence is what started the sender.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
||||
try:
|
||||
age = now - orphan.stat().st_mtime
|
||||
except OSError:
|
||||
continue
|
||||
if age > CLAIM_EXPIRY_SECONDS:
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
# Someone else holds a live lease on it. This check has to come
|
||||
# first. Claiming a file bumps its attempt count and refreshes its
|
||||
# mtime, so a sender that has just taken the final attempt looks
|
||||
# exhausted to everyone else while it is actively draining. Judging
|
||||
# exhaustion before liveness let a second sender unlink a batch out
|
||||
# from under its owner, losing every event in it.
|
||||
continue
|
||||
# Attempts, not age. Every re-claim touches the mtime and every release
|
||||
# backdates it by a fixed amount, so age is pinned near the stale
|
||||
# threshold and never reaches the expiry. Age stays only as a backstop
|
||||
# for files that never carried an attempt marker.
|
||||
if _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS or age > CLAIM_EXPIRY_SECONDS:
|
||||
try:
|
||||
orphan.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
continue
|
||||
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
||||
try:
|
||||
orphan.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _safe_mtime(path: Path) -> float:
|
||||
try:
|
||||
return path.stat().st_mtime
|
||||
except OSError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
||||
"""Persist the unsent remainder, atomically, and refresh the lease.
|
||||
|
||||
Called after every successful batch. Two jobs: a retry resumes where the
|
||||
send stopped instead of re-posting from the top, and the rewrite doubles as
|
||||
the lease heartbeat, so a slow sender does not have its claim stolen
|
||||
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
||||
"""
|
||||
if not remaining:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return True
|
||||
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
||||
try:
|
||||
payload = "".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining)
|
||||
# fsync before the rename: without it the rename can land while the
|
||||
# bytes have not, and the claim comes back empty or truncated after a
|
||||
# crash. _drain then reads zero events and unlinks it.
|
||||
with open(temporary, "w", encoding="utf-8") as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
temporary.replace(claim)
|
||||
_touch(claim)
|
||||
return True
|
||||
except OSError:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
||||
"""Persist the remainder and drop the lease, because this sender has given up.
|
||||
|
||||
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
||||
make an abandoned batch look actively owned for a further
|
||||
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
||||
threshold lets the next flush pick it up immediately, while the attempt
|
||||
count in the filename still bounds how many times that can happen.
|
||||
"""
|
||||
if not _rewrite_claim(claim, remaining):
|
||||
return
|
||||
try:
|
||||
# Backdate past the stale threshold so the next flush can pick it up,
|
||||
# minus a cooldown that grows with the attempts already spent. Clamped so
|
||||
# the mtime never lands in the future, which would read as a live lease.
|
||||
cooldown = min(_claim_attempt(claim) * RETRY_COOLDOWN_SECONDS, CLAIM_STALE_SECONDS)
|
||||
released = time.time() - CLAIM_STALE_SECONDS - 1 + cooldown
|
||||
os.utime(claim, (released, released))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_email(key: str) -> str:
|
||||
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
||||
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
||||
@@ -300,34 +800,130 @@ def _post(payload: dict[str, Any], url: str) -> bool:
|
||||
|
||||
|
||||
def resolve_distinct_id() -> tuple[str, str]:
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any."""
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
||||
|
||||
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
||||
anonymous id: aliasing one account email to another merges two real person
|
||||
profiles and cannot be undone, so a key that now belongs to a different
|
||||
account re-resolves with no alias.
|
||||
"""
|
||||
identity = _read_identity()
|
||||
email = identity.get("email", "")
|
||||
if email:
|
||||
return email, ""
|
||||
key = memory_core.api_key()
|
||||
fingerprint = _digest(key) if key else ""
|
||||
email = identity.get("email", "")
|
||||
|
||||
if email and fingerprint:
|
||||
recorded = identity.get("key_fingerprint", "")
|
||||
if recorded == fingerprint:
|
||||
return email, ""
|
||||
if not recorded:
|
||||
# Rows written before fingerprints existed. Verify rather than
|
||||
# adopt: a key changed before the upgrade would otherwise bind the
|
||||
# new key to the previous account's email, permanently, and the
|
||||
# fingerprint would then agree with itself forever after.
|
||||
verified = _resolve_email(key)
|
||||
if not verified:
|
||||
# Offline, firewalled, or the API is down. Keep the previous
|
||||
# behaviour and retry on the next flush rather than dropping a
|
||||
# real account attribution. Safe because the same network that
|
||||
# failed /v1/ping/ is about to fail the PostHog POST, so nothing
|
||||
# is delivered under the unverified identity in the meantime.
|
||||
return email, ""
|
||||
identity["email"] = verified
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return verified, ""
|
||||
|
||||
if not key:
|
||||
# No key to verify the account with; do not keep attributing to it.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
email = _resolve_email(key)
|
||||
if not email:
|
||||
|
||||
resolved = _resolve_email(key)
|
||||
if not resolved:
|
||||
# The key changed and will not resolve (revoked, offline, API down).
|
||||
# Reaching here with an email means the recorded fingerprint disagreed,
|
||||
# so the key really did change. Drop the account and rotate: the stored
|
||||
# anonymous id may already be merged into that account's person, and
|
||||
# reusing it would keep the events on the profile we are trying to
|
||||
# leave.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
previous = identity.get("anonymous_id", "")
|
||||
identity["email"] = email
|
||||
|
||||
# Alias only when going anonymous -> email for the first time. Once an anon
|
||||
# id has been merged into an account it must never be offered again: an
|
||||
# alias naming an already-identified id is what could link two real people.
|
||||
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
||||
if previous:
|
||||
identity["aliased"] = True
|
||||
identity["email"] = resolved
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return email, previous
|
||||
return resolved, previous
|
||||
|
||||
|
||||
def flush() -> int:
|
||||
"""Drain claimed spools to PostHog and return the number of events sent."""
|
||||
"""Drain the live spool, then any parked claims, and return events sent."""
|
||||
if not is_enabled():
|
||||
return 0
|
||||
claim = _claim_spool()
|
||||
sent, delivered = _drain(_claim_spool())
|
||||
if not delivered:
|
||||
# The network is failing. Retrying other batches now would only burn
|
||||
# their attempt budget against the same broken connection.
|
||||
return sent
|
||||
|
||||
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
||||
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
||||
directory = memory_core.data_dir()
|
||||
_sweep_debris(directory)
|
||||
for _ in range(MAX_PARKED_PER_RUN):
|
||||
parked = _claim_parked(directory)
|
||||
if parked is None:
|
||||
break
|
||||
count, delivered = _drain(parked)
|
||||
sent += count
|
||||
if not delivered:
|
||||
break
|
||||
return sent
|
||||
|
||||
|
||||
def _drain(claim: Path | None) -> tuple[int, bool]:
|
||||
"""Post one claimed batch file, recording progress after every batch.
|
||||
|
||||
Returns (events sent, whether everything was delivered).
|
||||
"""
|
||||
if claim is None:
|
||||
return 0
|
||||
return 0, True
|
||||
try:
|
||||
lines = claim.read_text(encoding="utf-8").splitlines()
|
||||
except ValueError:
|
||||
# UnicodeDecodeError from a torn write: the content is unrecoverable, so
|
||||
# quarantine rather than retry. flush() runs from a bare `finally:` in
|
||||
# flush_worker, so raising here also skips the handoff cleanup, and an
|
||||
# undecodable file would otherwise be re-read on every flush forever.
|
||||
# Reported as delivered because there is nothing left to deliver and the
|
||||
# rest of the run should continue.
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt"))
|
||||
except OSError:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0, True
|
||||
except OSError:
|
||||
return 0
|
||||
# Could not read it, which is not the same as having nothing to send.
|
||||
# The file is left exactly where it is: a vanished or briefly unreadable
|
||||
# claim is retryable, and quarantining it here would discard events over
|
||||
# a transient filesystem error. Reported as undelivered so the run stops
|
||||
# instead of counting a batch nothing was posted from as delivered.
|
||||
return 0, False
|
||||
events = []
|
||||
for line in lines:
|
||||
try:
|
||||
@@ -337,11 +933,18 @@ def flush() -> int:
|
||||
if isinstance(value, dict) and value.get("event"):
|
||||
events.append(value)
|
||||
if not events:
|
||||
# Only delete when the file really is empty. A non-empty file that
|
||||
# parses to nothing is a torn write, and its contents are the unsent
|
||||
# remainder — deleting it is the data loss this PR exists to prevent.
|
||||
try:
|
||||
claim.unlink()
|
||||
empty = claim.stat().st_size == 0
|
||||
except OSError:
|
||||
empty = True
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt")) if not empty else claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
return 0, True
|
||||
|
||||
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
||||
if aliased_anonymous_id:
|
||||
@@ -360,12 +963,17 @@ def flush() -> int:
|
||||
|
||||
sent = 0
|
||||
for start in range(0, len(events), BATCH_SIZE):
|
||||
chunk = events[start : start + BATCH_SIZE]
|
||||
batch = [
|
||||
{
|
||||
"event": event["event"],
|
||||
"distinct_id": distinct_id,
|
||||
# Carried through from record() so a resend can be collapsed.
|
||||
"uuid": event.get("uuid"),
|
||||
"timestamp": event.get("timestamp"),
|
||||
"properties": {
|
||||
# Fallback only: events recorded by a build before source
|
||||
# moved into record() have none of their own.
|
||||
"source": _source_tag,
|
||||
"language": "python",
|
||||
"$process_person_profile": False,
|
||||
@@ -373,16 +981,24 @@ def flush() -> int:
|
||||
**(event.get("properties") or {}),
|
||||
},
|
||||
}
|
||||
for event in events[start : start + BATCH_SIZE]
|
||||
for event in chunk
|
||||
]
|
||||
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
||||
return sent
|
||||
sent += len(batch)
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return sent
|
||||
# Keep only what has not been delivered, and release the lease.
|
||||
# Previously the whole file was kept and the retry re-posted every
|
||||
# batch, including the ones that had already arrived.
|
||||
_release_claim(claim, events[start:])
|
||||
return sent, False
|
||||
sent += len(chunk)
|
||||
# Record progress and refresh the lease after each successful batch, so
|
||||
# a crash repeats at most one batch instead of the entire file. If the
|
||||
# rewrite fails the claim still holds delivered events, so stop rather
|
||||
# than carry on as though progress were recorded — continuing is how the
|
||||
# duplicate delivery this PR fixes would come back.
|
||||
if not _rewrite_claim(claim, events[start + len(chunk) :]):
|
||||
_release_claim(claim, events[start + len(chunk) :])
|
||||
return sent, False
|
||||
return sent, True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"id": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"homepage": "https://docs.mem0.ai/integrations/claude-code",
|
||||
"native": {
|
||||
"pluginRoot": "${CLAUDE_PLUGIN_ROOT}",
|
||||
|
||||
@@ -7,8 +7,8 @@ disable-model-invocation: true
|
||||
# Pause memory capture
|
||||
|
||||
To pause (hooks stop capturing and sending session content; a minimal
|
||||
anonymous telemetry ping still fires at session start unless
|
||||
`MEM0_TELEMETRY=false`):
|
||||
telemetry ping still fires at session start, under your Mem0 account email,
|
||||
unless `MEM0_TELEMETRY=false`):
|
||||
|
||||
```bash
|
||||
python3 "${CLAUDE_PLUGIN_ROOT}/core/memory_cli.py" --harness "claude-code" --plugin-data-dir "${CLAUDE_PLUGIN_DATA}" pause
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import os
|
||||
import sqlite3
|
||||
import subprocess
|
||||
@@ -3460,7 +3461,12 @@ def test_automatic_flush_can_be_disabled_for_external_harnesses(isolated_env):
|
||||
def test_version_is_single_sourced():
|
||||
manifest = json.loads((PLUGIN_ROOT / ".claude-plugin" / "plugin.json").read_text())
|
||||
assert manifest["name"] == "mem0"
|
||||
assert manifest["version"] == memory_core.PLUGIN_VERSION == "0.3.1"
|
||||
# Compared against PLUGIN_VERSION, never a literal. A hardcoded version here
|
||||
# was one more place to edit on every release, inside the test asserting the
|
||||
# version is single-sourced, and it caught nothing that the agreement checks
|
||||
# below do not: fifteen places set to the same wrong value would still pass.
|
||||
assert re.fullmatch(r"\d+\.\d+\.\d+", memory_core.PLUGIN_VERSION), memory_core.PLUGIN_VERSION
|
||||
assert manifest["version"] == memory_core.PLUGIN_VERSION
|
||||
root = REPOSITORY_ROOT
|
||||
for mp in (root / "marketplace.json", root / ".claude-plugin" / "marketplace.json"):
|
||||
entry = next(p for p in json.loads(mp.read_text())["plugins"] if p["name"] == "mem0")
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
@@ -198,9 +199,11 @@ def test_a_stale_claim_is_reclaimed(isolated_env, monkeypatch):
|
||||
telemetry.record("search")
|
||||
orphan = telemetry._claim_spool()
|
||||
assert orphan is not None
|
||||
monkeypatch.setattr(
|
||||
telemetry.time, "time", lambda: orphan.stat().st_mtime + telemetry.CLAIM_STALE_SECONDS + 1
|
||||
)
|
||||
# Frozen rather than re-stat'd per call: flush() drains the live spool and
|
||||
# then looks for parked claims in the same run, so by the second look this
|
||||
# file no longer exists.
|
||||
stale_now = orphan.stat().st_mtime + telemetry.CLAIM_STALE_SECONDS + 1
|
||||
monkeypatch.setattr(telemetry.time, "time", lambda: stale_now)
|
||||
|
||||
with patch.object(telemetry, "_post", lambda payload, url: True):
|
||||
assert telemetry.flush() == 1
|
||||
@@ -210,9 +213,13 @@ def test_an_expired_claim_is_dropped(isolated_env, monkeypatch):
|
||||
telemetry.record("search")
|
||||
orphan = telemetry._claim_spool()
|
||||
assert orphan is not None
|
||||
monkeypatch.setattr(
|
||||
telemetry.time, "time", lambda: orphan.stat().st_mtime + telemetry.CLAIM_EXPIRY_SECONDS + 1
|
||||
)
|
||||
expired_now = orphan.stat().st_mtime + telemetry.CLAIM_EXPIRY_SECONDS + 1
|
||||
monkeypatch.setattr(telemetry.time, "time", lambda: expired_now)
|
||||
# Expiry now only discards a batch that was genuinely retried and failed,
|
||||
# so age alone is not enough — age it past the attempt budget too.
|
||||
retried = orphan.parent / orphan.name.replace("-a0.", f"-a{telemetry.MAX_CLAIM_ATTEMPTS}.")
|
||||
orphan.replace(retried)
|
||||
os.utime(retried, (expired_now, expired_now - telemetry.CLAIM_EXPIRY_SECONDS - 1))
|
||||
assert telemetry._claim_spool() is None
|
||||
assert not list(memory_core.data_dir().glob("telemetry-*.sending"))
|
||||
|
||||
@@ -258,12 +265,150 @@ def test_an_unresolvable_key_falls_back_to_the_anonymous_id(isolated_env, monkey
|
||||
assert telemetry.resolve_distinct_id()[0].startswith("code-anon-")
|
||||
|
||||
|
||||
def test_is_first_run_flips_after_the_first_identity_write(isolated_env):
|
||||
def test_logging_out_does_not_leave_events_on_the_previous_account(isolated_env, monkeypatch):
|
||||
"""Review finding: clearing the email kept an id already merged into a person.
|
||||
|
||||
The anonymous id is offered to PostHog as $anon_distinct_id on first sign-in,
|
||||
and that merge is permanent. Keeping it after the key goes away means every
|
||||
later anonymous event lands on the account that just left.
|
||||
"""
|
||||
# Run anonymously first, which is the only way an id exists to be merged.
|
||||
merged = telemetry.anonymous_id()
|
||||
|
||||
monkeypatch.setenv("MEM0_API_KEY", "key-for-account-a")
|
||||
with patch.object(telemetry, "_resolve_email", lambda key: "a@example.com"):
|
||||
identified, alias = telemetry.resolve_distinct_id()
|
||||
assert identified == "a@example.com"
|
||||
assert alias == merged, "the anonymous id was merged into this account"
|
||||
|
||||
monkeypatch.delenv("MEM0_API_KEY", raising=False)
|
||||
after_logout, logout_alias = telemetry.resolve_distinct_id()
|
||||
|
||||
assert after_logout.startswith("code-anon-")
|
||||
assert after_logout != merged, "reused an id already merged into the previous account"
|
||||
assert logout_alias == ""
|
||||
assert "aliased" not in telemetry._read_identity(), "rotated id must be aliasable again"
|
||||
|
||||
|
||||
def test_a_changed_key_that_will_not_resolve_rotates_the_anonymous_id(isolated_env, monkeypatch):
|
||||
"""Same leak by the other route: fingerprint disagrees and the lookup fails."""
|
||||
merged = telemetry.anonymous_id()
|
||||
monkeypatch.setenv("MEM0_API_KEY", "key-for-account-a")
|
||||
with patch.object(telemetry, "_resolve_email", lambda key: "a@example.com"):
|
||||
telemetry.resolve_distinct_id()
|
||||
|
||||
monkeypatch.setenv("MEM0_API_KEY", "key-for-account-b")
|
||||
with patch.object(telemetry, "_resolve_email", lambda key: ""):
|
||||
after, alias = telemetry.resolve_distinct_id()
|
||||
|
||||
assert after.startswith("code-anon-")
|
||||
assert after != merged
|
||||
assert alias == ""
|
||||
assert "email" not in telemetry._read_identity()
|
||||
|
||||
|
||||
def test_a_legacy_cached_email_is_verified_before_the_key_is_bound(isolated_env, monkeypatch):
|
||||
"""Review finding: a key changed before upgrading bound the wrong account.
|
||||
|
||||
Rows written before fingerprints existed carry an email and no fingerprint.
|
||||
Adopting the current key without checking pinned that key to the previous
|
||||
account's email, and every run after that agreed with itself.
|
||||
"""
|
||||
telemetry._write_identity({"email": "old@example.com", "anonymous_id": "code-anon-seed"})
|
||||
monkeypatch.setenv("MEM0_API_KEY", "key-for-account-b")
|
||||
|
||||
with patch.object(telemetry, "_resolve_email", lambda key: "new@example.com"):
|
||||
resolved, alias = telemetry.resolve_distinct_id()
|
||||
|
||||
assert resolved == "new@example.com"
|
||||
assert alias == "", "email to email must never alias; it merges two real people"
|
||||
stored = telemetry._read_identity()
|
||||
assert stored["email"] == "new@example.com"
|
||||
assert stored["key_fingerprint"] == telemetry._digest("key-for-account-b")
|
||||
|
||||
|
||||
def test_a_legacy_row_keeps_working_when_the_account_cannot_be_checked(isolated_env, monkeypatch):
|
||||
"""Firewalled users must not lose attribution, and must not bind unverified.
|
||||
|
||||
The same network that fails /v1/ping/ fails the PostHog POST, so nothing is
|
||||
delivered under the unverified identity while this holds.
|
||||
"""
|
||||
telemetry._write_identity({"email": "old@example.com"})
|
||||
monkeypatch.setenv("MEM0_API_KEY", "key-for-account-b")
|
||||
|
||||
with patch.object(telemetry, "_resolve_email", lambda key: ""):
|
||||
resolved, _ = telemetry.resolve_distinct_id()
|
||||
|
||||
assert resolved == "old@example.com"
|
||||
assert "key_fingerprint" not in telemetry._read_identity(), "bound an unverified key"
|
||||
|
||||
|
||||
def test_a_failed_upgrade_claim_can_be_retried(isolated_env, monkeypatch):
|
||||
"""Review finding: a failed rewrite left the sentinel and suppressed forever.
|
||||
|
||||
claim_version_change returns early on FileExistsError, and the marker still
|
||||
holds the old version, so the upgrade for that version was never recorded
|
||||
again on that machine.
|
||||
"""
|
||||
telemetry.claim_install()
|
||||
state_path = memory_core.data_dir() / "install-state.json"
|
||||
state = json.loads(state_path.read_text())
|
||||
state["plugin_version"] = "0.0.1-old"
|
||||
state_path.write_text(json.dumps(state), encoding="utf-8")
|
||||
|
||||
real_replace = Path.replace
|
||||
|
||||
def failing_replace(self, target):
|
||||
raise OSError("disk full")
|
||||
|
||||
monkeypatch.setattr(Path, "replace", failing_replace)
|
||||
assert telemetry.claim_version_change() is None
|
||||
|
||||
monkeypatch.setattr(Path, "replace", real_replace)
|
||||
assert telemetry.claim_version_change() == "0.0.1-old", "sentinel suppressed the retry"
|
||||
|
||||
|
||||
def test_first_run_is_not_flipped_by_writing_the_identity_file(isolated_env):
|
||||
"""The identity file is written by a successful flush, not by recording.
|
||||
|
||||
Keying first-run off it meant an offline user recorded code.install on every
|
||||
session forever, and every 0.2.x user recorded one on their first 0.3.x run.
|
||||
"""
|
||||
assert telemetry.is_first_run()
|
||||
telemetry.anonymous_id()
|
||||
assert telemetry.is_first_run()
|
||||
|
||||
|
||||
def test_claiming_install_ends_first_run(isolated_env):
|
||||
assert telemetry.claim_install() == "install"
|
||||
assert not telemetry.is_first_run()
|
||||
|
||||
|
||||
def test_install_can_only_be_claimed_once(isolated_env):
|
||||
"""Two sessions starting together must not both record an install."""
|
||||
assert telemetry.claim_install() == "install"
|
||||
assert telemetry.claim_install() is None
|
||||
|
||||
|
||||
def test_a_populated_data_dir_reads_as_an_upgrade(isolated_env):
|
||||
"""A fresh install has an empty data directory; anything else predates it."""
|
||||
data_dir = memory_core.data_dir()
|
||||
data_dir.mkdir(parents=True, exist_ok=True)
|
||||
(data_dir / "requirements.txt").write_text("mem0ai\n", encoding="utf-8")
|
||||
assert telemetry.claim_install() == "upgrade"
|
||||
|
||||
|
||||
def test_a_version_change_is_claimed_once(isolated_env):
|
||||
telemetry.claim_install()
|
||||
state_path = memory_core.data_dir() / "install-state.json"
|
||||
state = json.loads(state_path.read_text())
|
||||
state["plugin_version"] = "0.0.1-old"
|
||||
state_path.write_text(json.dumps(state), encoding="utf-8")
|
||||
|
||||
assert telemetry.claim_version_change() == "0.0.1-old"
|
||||
assert telemetry.claim_version_change() is None
|
||||
|
||||
|
||||
def test_spawn_flush_does_nothing_without_a_spool(isolated_env):
|
||||
with patch.object(telemetry.subprocess, "Popen") as popen:
|
||||
assert telemetry.spawn_flush() is False
|
||||
@@ -273,3 +418,114 @@ def test_spawn_flush_does_nothing_without_a_spool(isolated_env):
|
||||
with patch.object(telemetry.subprocess, "Popen") as popen:
|
||||
assert telemetry.spawn_flush() is True
|
||||
popen.assert_called_once()
|
||||
|
||||
|
||||
def test_salt_is_stable_across_processes(isolated_env):
|
||||
"""Hooks are separate short-lived processes; one repo must hash one way.
|
||||
|
||||
An unlocked read-modify-write let each process mint its own salt, so a
|
||||
repository hashed several ways in the window before one writer won.
|
||||
"""
|
||||
import subprocess as sp
|
||||
|
||||
core = str(Path(__file__).resolve().parents[1] / "core")
|
||||
script = (
|
||||
f"import sys; sys.path.insert(0, {core!r})\n"
|
||||
"import telemetry\n"
|
||||
"print(telemetry._install_salt())"
|
||||
)
|
||||
env = {**os.environ, "MEM0_CODE_DATA_DIR": str(memory_core.data_dir())}
|
||||
salts = {
|
||||
sp.run([sys.executable, "-c", script], capture_output=True, text=True, env=env).stdout.strip()
|
||||
for _ in range(4)
|
||||
}
|
||||
assert len(salts) == 1, f"one repo hashed {len(salts)} ways: {salts}"
|
||||
|
||||
|
||||
def test_salt_does_not_touch_the_identity_file(isolated_env):
|
||||
"""The identity file is is_first_run's marker and the sender's email store.
|
||||
|
||||
Writing the salt into it would create it from record(), suppressing the
|
||||
install event, and would race resolve_distinct_id, which holds a stale copy
|
||||
of that dict across a network call.
|
||||
"""
|
||||
telemetry._install_salt()
|
||||
assert not telemetry._identity_path().exists()
|
||||
|
||||
|
||||
def test_no_salt_means_no_hash_rather_than_an_unsalted_one(isolated_env, monkeypatch):
|
||||
"""A read-only data dir drops the property; it must not emit a weak digest.
|
||||
|
||||
The previous fallback was a digest of the salt file's own path, which an
|
||||
attacker can compute, memoized for the whole process. A property named
|
||||
repo_hash carrying an effectively unsalted digest is worse than no property:
|
||||
it reads as protected and is not.
|
||||
"""
|
||||
telemetry._salt_cache = ""
|
||||
monkeypatch.setattr(telemetry.os, "open", lambda *a, **k: (_ for _ in ()).throw(OSError("read-only")))
|
||||
|
||||
assert telemetry._install_salt() == ""
|
||||
assert telemetry._scoped_digest("git@github.com:acme/secret.git") == ""
|
||||
|
||||
|
||||
def test_a_half_written_salt_is_never_visible_to_another_process(isolated_env, monkeypatch):
|
||||
"""The window this closes: file created, value not yet written.
|
||||
|
||||
O_CREAT|O_EXCL then write leaves the name present and empty in between. A
|
||||
hook reading it there used to get "", fall back to the path digest and cache
|
||||
that for its whole run, so the same repo hashed two ways depending on timing.
|
||||
Publishing by link means the name either does not exist or is complete.
|
||||
"""
|
||||
telemetry._salt_cache = ""
|
||||
salt_path = telemetry._salt_path()
|
||||
observed = []
|
||||
|
||||
real_link = telemetry.os.link
|
||||
|
||||
def observing_link(source, target):
|
||||
# Stand where the racing reader stands: after the temp file is written,
|
||||
# before the real name exists.
|
||||
observed.append(salt_path.exists())
|
||||
return real_link(source, target)
|
||||
|
||||
monkeypatch.setattr(telemetry.os, "link", observing_link)
|
||||
salt = telemetry._install_salt()
|
||||
|
||||
assert observed == [False], "the salt name existed before it held a value"
|
||||
assert len(salt) == 32
|
||||
assert salt_path.read_text(encoding="utf-8").strip() == salt
|
||||
|
||||
|
||||
def test_a_filesystem_without_hardlinks_still_gets_a_salt(isolated_env, monkeypatch):
|
||||
"""Publishing by link must not become a silent loss of the hashes.
|
||||
|
||||
Some network mounts and container volumes reject os.link. Returning ""
|
||||
there would drop repo_hash and session_hash on every run for that whole
|
||||
cohort, which is a bigger loss than the narrow race the link closes.
|
||||
"""
|
||||
telemetry._salt_cache = ""
|
||||
monkeypatch.setattr(
|
||||
telemetry.os, "link", lambda src, dst: (_ for _ in ()).throw(OSError(38, "not implemented"))
|
||||
)
|
||||
|
||||
salt = telemetry._install_salt()
|
||||
|
||||
assert len(salt) == 32, "no salt on a filesystem without hardlinks"
|
||||
assert telemetry._salt_path().read_text(encoding="utf-8").strip() == salt
|
||||
assert telemetry._scoped_digest("git@github.com:acme/x.git") != ""
|
||||
assert not list(telemetry._salt_path().parent.glob("telemetry-salt.*.tmp"))
|
||||
|
||||
|
||||
def test_a_concurrent_writer_does_not_clobber_the_published_salt(isolated_env):
|
||||
"""Second process to finish must adopt the first one's salt, not replace it.
|
||||
|
||||
os.link rather than os.replace is what makes losing the race harmless.
|
||||
"""
|
||||
telemetry._salt_cache = ""
|
||||
first = telemetry._install_salt()
|
||||
|
||||
telemetry._salt_cache = ""
|
||||
second = telemetry._install_salt()
|
||||
|
||||
assert second == first
|
||||
assert not list(telemetry._salt_path().parent.glob("telemetry-salt.*.tmp")), "temp file left behind"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"author": { "name": "Mem0", "email": "support@mem0.ai" },
|
||||
"homepage": "https://docs.mem0.ai/integrations/codex",
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Generated by integrations/agent-plugin-core/build/build.py. Do not edit."""
|
||||
|
||||
HARNESS_ID = "codex"
|
||||
SOURCE_TAG = "CODEX_PLUGIN"
|
||||
|
||||
# Platform-side vocabulary (mem0_event.source + X-Application). The whole
|
||||
# plugin family is one source; which editor it runs in is the application.
|
||||
# An empty application means the host is unknown, and memory_core omits
|
||||
# the header entirely rather than sending a placeholder.
|
||||
PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
PLATFORM_APPLICATION = "codex"
|
||||
@@ -290,6 +290,11 @@ def run(
|
||||
if args.plugin_data_dir:
|
||||
os.environ[data_dir_env] = args.plugin_data_dir
|
||||
|
||||
# Snapshot BEFORE anything writes to the data dir: cache_plugin_api_key
|
||||
# writes `api-key` and EvidenceStore creates `evidence.sqlite3`, so asking
|
||||
# after them always saw content and every fresh install reported an upgrade.
|
||||
data_dir_was_empty = telemetry.data_dir_was_empty()
|
||||
|
||||
cache_plugin_api_key()
|
||||
if args.action == "session-start":
|
||||
clear_stale_api_key_cache()
|
||||
@@ -305,8 +310,19 @@ def run(
|
||||
return 0
|
||||
|
||||
if args.action == "session-start":
|
||||
if telemetry.is_first_run():
|
||||
# Claims the marker atomically and says which event to record, so a
|
||||
# second session starting alongside this one cannot record it too.
|
||||
first_event = telemetry.claim_install(was_empty=data_dir_was_empty)
|
||||
if first_event == "install":
|
||||
telemetry.record("install")
|
||||
elif first_event == "upgrade":
|
||||
# First run after a build that never wrote the marker; the
|
||||
# predecessor version was never recorded anywhere.
|
||||
telemetry.record("upgrade", from_version="pre-0.3")
|
||||
else:
|
||||
previous = telemetry.claim_version_change()
|
||||
if previous:
|
||||
telemetry.record("upgrade", from_version=previous)
|
||||
recovered = recover_pending_handoffs()
|
||||
record_session_start(store, hook_input)
|
||||
if recovered:
|
||||
|
||||
@@ -29,7 +29,7 @@ from typing import Any, Iterable
|
||||
import telemetry
|
||||
|
||||
DEFAULT_API_URL = "https://api.mem0.ai"
|
||||
PLUGIN_VERSION = "0.3.1"
|
||||
PLUGIN_VERSION = "0.3.2"
|
||||
|
||||
_harness_name: str = "generic"
|
||||
_harness_env_prefix: str = "MEM0_PLUGIN"
|
||||
@@ -1800,6 +1800,34 @@ def extraction_message_batches(
|
||||
return batches
|
||||
|
||||
|
||||
# Platform surface attribution. Read from the generated per-host module so a new
|
||||
# entrypoint is correct without remembering to configure anything.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
except ImportError:
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
|
||||
def platform_headers(key: str) -> dict[str, str]:
|
||||
"""Auth plus the three surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are set-once by contract: this is the
|
||||
outermost layer, so it sets them, and nothing below may overwrite them.
|
||||
X-Mem0-Client is append-only — anything downstream adds itself to the tail.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": _PLATFORM_SOURCE,
|
||||
"X-Mem0-Client": f"mem0-plugin/{PLUGIN_VERSION}",
|
||||
}
|
||||
if _PLATFORM_APPLICATION:
|
||||
headers["X-Application"] = _PLATFORM_APPLICATION
|
||||
return headers
|
||||
|
||||
|
||||
def _request_json(
|
||||
url: str, key: str, payload: dict[str, Any], timeout: float
|
||||
) -> tuple[dict[str, Any] | list[Any], int, int]:
|
||||
@@ -1807,7 +1835,7 @@ def _request_json(
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1834,7 +1862,7 @@ def _get_json(
|
||||
) -> tuple[dict[str, Any] | list[Any], int]:
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="GET",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1980,6 +2008,13 @@ def flush_session(
|
||||
"user_id": write_user,
|
||||
"app_id": repo.app_id,
|
||||
"run_id": session_id,
|
||||
# Top level, not metadata: the backend reads `source` from the body or
|
||||
# the query string, never from metadata, which is where this used to
|
||||
# sit. The X-Mem0-Source header is also read, but only from the
|
||||
# platform release that ships alongside this change, so the body value
|
||||
# is what makes attribution work on both. The harness tag stays in
|
||||
# metadata as hook provenance.
|
||||
"source": _PLATFORM_SOURCE,
|
||||
"metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)},
|
||||
"agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS,
|
||||
"custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS,
|
||||
@@ -2523,7 +2558,7 @@ def _collect_memory_ids(
|
||||
def _delete_memory(api_url: str, key: str, memory_id: str) -> bool:
|
||||
request = urllib.request.Request(
|
||||
f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/",
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Anonymous usage telemetry for Mem0 agent plugins.
|
||||
"""Usage telemetry for Mem0 agent plugins.
|
||||
|
||||
Events are linked to your Mem0 account email when an API key is configured, and
|
||||
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
||||
attribute the same way.
|
||||
|
||||
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
||||
touches the network: `record` appends one JSON line to a local spool and returns.
|
||||
@@ -9,7 +13,8 @@ started once per session and again from the flush worker that is already detache
|
||||
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
||||
|
||||
Never sends prompts, memory text, queries, file paths, repository names, or API
|
||||
keys: only event names, durations, counts, coarse outcomes, and salted hashes.
|
||||
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
||||
identifiers hashed with a random per-install salt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,8 +34,24 @@ from typing import Any
|
||||
|
||||
import memory_core
|
||||
|
||||
_harness: str = "generic"
|
||||
_source_tag: str = "MEM0_PLUGIN"
|
||||
# Seeded from the per-host module the build generates into core/. Two processes
|
||||
# in this pipeline never call init() — mcp_server.py, and the detached
|
||||
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
||||
# was what every one of their events got labelled with.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
||||
except ImportError:
|
||||
_DEFAULT_HARNESS = "generic"
|
||||
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
_salt_cache: str = ""
|
||||
_harness: str = _DEFAULT_HARNESS
|
||||
_source_tag: str = _DEFAULT_SOURCE_TAG
|
||||
_PRIVATE_KEYS = {
|
||||
"apikey",
|
||||
"authorization",
|
||||
@@ -56,10 +77,19 @@ _PRIVATE_KEYS = {
|
||||
}
|
||||
|
||||
|
||||
def init(harness: str = "generic", source_tag: str = "") -> None:
|
||||
def init(harness: str = "", source_tag: str = "") -> None:
|
||||
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
||||
|
||||
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
||||
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
||||
meant one plugin could emit three different source values depending on which
|
||||
process happened to send the batch.
|
||||
"""
|
||||
global _harness, _source_tag
|
||||
_harness = harness
|
||||
_source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN"
|
||||
_harness = harness or _DEFAULT_HARNESS
|
||||
_source_tag = source_tag or (
|
||||
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
||||
)
|
||||
|
||||
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
||||
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
||||
@@ -70,6 +100,16 @@ BATCH_SIZE = 100
|
||||
SEND_TIMEOUT = 5
|
||||
CLAIM_STALE_SECONDS = 120
|
||||
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
||||
# A batch is only discarded once it has genuinely been retried this many times.
|
||||
MAX_CLAIM_ATTEMPTS = 3
|
||||
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
||||
# cannot turn one flush into an unbounded send loop.
|
||||
MAX_PARKED_PER_RUN = 3
|
||||
# Added to the wait before a released claim becomes reclaimable, per attempt
|
||||
# already spent. Releasing straight to "reclaimable now" let two senders burn the
|
||||
# whole budget within seconds of one another on a single momentary failure, and
|
||||
# discard a batch a retry a minute later would have delivered.
|
||||
RETRY_COOLDOWN_SECONDS = 60
|
||||
|
||||
|
||||
def is_enabled() -> bool:
|
||||
@@ -83,9 +123,126 @@ def is_enabled() -> bool:
|
||||
|
||||
|
||||
def _digest(value: str, length: int = 16) -> str:
|
||||
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _salt_path() -> Path:
|
||||
return memory_core.data_dir() / "telemetry-salt"
|
||||
|
||||
|
||||
def _install_salt() -> str:
|
||||
"""Random per-install salt, created once and memoized for the process.
|
||||
|
||||
Deliberately its own file, claimed with O_CREAT|O_EXCL, rather than a key in
|
||||
the identity file. Three reasons, all of which produced wrong data when this
|
||||
lived in the identity dict:
|
||||
|
||||
- Hooks are short-lived separate processes firing on every tool call, and
|
||||
people run more than one agent window. A read-modify-write would let each
|
||||
process mint its own salt, so one repository would hash several ways in the
|
||||
window before a writer won.
|
||||
- resolve_distinct_id holds a copy of the identity dict across a network call
|
||||
to /v1/ping/, so whichever write landed second erased the other's key —
|
||||
losing either the salt (repo_hash changes mid-stream) or the email (a
|
||||
second $identify, splitting the person).
|
||||
- Touching the identity file from record() would create it, and is_first_run
|
||||
keys off that file, so recording an event would silently suppress the
|
||||
install event.
|
||||
|
||||
Published atomically, and there is deliberately no derived fallback. Creating
|
||||
the file with O_CREAT|O_EXCL and then writing into it leaves a window where
|
||||
the file exists and is empty, and a concurrent hook that reads it in that
|
||||
window gets nothing. Falling back to a digest of the path would hand that
|
||||
process a salt an attacker can compute, memoized for its whole run, which is
|
||||
the privacy control this function exists to provide silently turning itself
|
||||
off under load. The salt is written to a private temp file first and linked
|
||||
into place, so the name either does not exist or already has the full value.
|
||||
|
||||
Returns "" when it genuinely cannot persist. Callers omit the hash entirely
|
||||
rather than emit an unsalted one.
|
||||
"""
|
||||
global _salt_cache
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
|
||||
path = _salt_path()
|
||||
# Read before writing. Hooks are separate processes firing on every tool
|
||||
# call, so all but the first find the salt already published; going straight
|
||||
# to create-fsync-link-unlink meant every one of them paid an fsync to
|
||||
# discover that, on a path whose whole promise is appending a line and
|
||||
# returning.
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
temporary = path.with_name(f"{path.name}.{os.getpid()}.tmp")
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
stream.write(uuid.uuid4().hex)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
try:
|
||||
# Atomic claim: fails if another process already published one.
|
||||
# os.link rather than replace, which would clobber theirs.
|
||||
os.link(temporary, path)
|
||||
except FileExistsError:
|
||||
pass
|
||||
except OSError:
|
||||
# No hardlinks here (some network mounts, some container volumes).
|
||||
# Claim the name directly instead. That reopens the empty-file
|
||||
# window, but the window is now benign: a reader that lands in it
|
||||
# gets "" and omits the hash for that process rather than caching a
|
||||
# guessable one. Losing the hashes on every run of an entire
|
||||
# filesystem is the worse failure.
|
||||
try:
|
||||
fallback = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fallback, "w", encoding="utf-8") as stream:
|
||||
stream.write(temporary.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_salt_cache = ""
|
||||
return _salt_cache
|
||||
|
||||
|
||||
def _scoped_digest(value: str, length: int = 16) -> str:
|
||||
"""Salted digest for values drawn from a guessable space.
|
||||
|
||||
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
||||
no remote — which normally contains the account username. Sixteen unsalted
|
||||
hex characters over that input space is enumerable, so this is not a
|
||||
privacy control without the salt. Salting per install keeps every
|
||||
within-account join the analytics actually use and gives up only
|
||||
cross-machine joins on the same repository, which nothing computes.
|
||||
|
||||
Returns "" when there is no salt, so record() omits the property. An
|
||||
unsalted digest over this input space is close to plaintext, and emitting one
|
||||
under a name that implies it is hashed is worse than sending nothing.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
salt = _install_salt()
|
||||
if not salt:
|
||||
return ""
|
||||
return hashlib.sha256(f"{salt}:{value}".encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _safe_value(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return memory_core.redact(value)
|
||||
@@ -145,9 +302,176 @@ def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
||||
return created
|
||||
|
||||
|
||||
def _rotate_anonymous_id(identity: dict[str, str]) -> str:
|
||||
"""Mint a fresh anonymous id because the account context is gone.
|
||||
|
||||
The previous id may already have been merged into a person profile by an
|
||||
$identify, and that merge is permanent. Reusing it after a logout or a key
|
||||
change attributes everything that follows to the account that just went
|
||||
away, which is the same misattribution the key fingerprint exists to stop,
|
||||
only arriving through the anonymous path instead.
|
||||
|
||||
`aliased` is cleared with it: the new id has never been merged, so it is
|
||||
eligible to be aliased into whatever account comes next.
|
||||
"""
|
||||
created = f"code-anon-{uuid.uuid4().hex}"
|
||||
identity["anonymous_id"] = created
|
||||
identity.pop("aliased", None)
|
||||
_write_identity(identity)
|
||||
return created
|
||||
|
||||
|
||||
def _install_state_path() -> Path:
|
||||
return memory_core.data_dir() / "install-state.json"
|
||||
|
||||
|
||||
def is_first_run() -> bool:
|
||||
"""Whether this machine has never recorded a plugin event before."""
|
||||
return not _identity_path().exists()
|
||||
"""Whether install has never been recorded on this machine.
|
||||
|
||||
Deliberately NOT the identity file. That file is only written by a
|
||||
successful flush, so an offline or firewalled user recorded code.install on
|
||||
every single session, forever — and every 0.2.x user recorded one on their
|
||||
first 0.3.x session because 0.2.x never wrote it at all.
|
||||
"""
|
||||
return not _install_state_path().exists()
|
||||
|
||||
|
||||
def data_dir_was_empty() -> bool:
|
||||
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
||||
|
||||
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
||||
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
||||
asking at claim time always saw content and every fresh install reported an
|
||||
upgrade. The caller snapshots this at the top of the run instead.
|
||||
"""
|
||||
return not _data_dir_has_content()
|
||||
|
||||
|
||||
def claim_install(was_empty: bool | None = None) -> str | None:
|
||||
"""Claim the one install/upgrade record for this machine, atomically.
|
||||
|
||||
Returns the event to record ("install" or "upgrade"), or None if another
|
||||
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
||||
cannot both win.
|
||||
|
||||
`was_empty` must come from data_dir_was_empty() called before this process
|
||||
wrote anything. Omitting it falls back to checking now, which is only
|
||||
correct for a caller that has touched nothing.
|
||||
"""
|
||||
if not is_enabled():
|
||||
# Never consume the one-shot claim while the user is opted out, or they
|
||||
# would silently lose their install event if they later opt in.
|
||||
return None
|
||||
|
||||
path = _install_state_path()
|
||||
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
json.dump(
|
||||
{
|
||||
"plugin_version": memory_core.PLUGIN_VERSION,
|
||||
"installed_at": memory_core.utc_now(),
|
||||
"upgraded": upgrading,
|
||||
},
|
||||
stream,
|
||||
)
|
||||
# Durable before this returns. The O_EXCL open is what makes the
|
||||
# claim exclusive, so it cannot be replaced by a temp-and-rename
|
||||
# without losing that, which leaves the content as the thing to make
|
||||
# safe. A kill between the open and this fsync used to leave a marker
|
||||
# that exists but parses to nothing: is_first_run reads it as claimed
|
||||
# and claim_version_change cannot read a version out of it.
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
except OSError:
|
||||
pass
|
||||
return "upgrade" if upgrading else "install"
|
||||
|
||||
|
||||
def _data_dir_has_content() -> bool:
|
||||
"""Whether anything predates this session in the plugin data directory."""
|
||||
try:
|
||||
for entry in memory_core.data_dir().iterdir():
|
||||
if entry.name != "install-state.json":
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _repair_install_state(path: Path) -> None:
|
||||
"""Rewrite an unparseable marker so version tracking can resume."""
|
||||
try:
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
temporary.write_text(
|
||||
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def claim_version_change() -> str | None:
|
||||
"""Return the previously recorded version if it differs, updating the marker.
|
||||
|
||||
Only meaningful once the marker exists — the first transition into 0.3.x has
|
||||
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
||||
the marker means the next session sees no change and records nothing.
|
||||
"""
|
||||
path = _install_state_path()
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
return None
|
||||
except json.JSONDecodeError:
|
||||
# A crash between O_EXCL and the write leaves an empty marker. Left
|
||||
# alone it disables every future upgrade event on this machine, because
|
||||
# claim_install sees the file and this function cannot parse it.
|
||||
state = None
|
||||
if not isinstance(state, dict):
|
||||
_repair_install_state(path)
|
||||
return None
|
||||
previous = str(state.get("plugin_version") or "")
|
||||
if not previous or previous == memory_core.PLUGIN_VERSION:
|
||||
return None
|
||||
# Claim the transition with an exclusive sentinel before rewriting the
|
||||
# marker. A plain read-modify-write let every concurrently starting session
|
||||
# observe the old version and each record its own upgrade — and the first
|
||||
# session after a version bump is exactly when several agent windows restart
|
||||
# together.
|
||||
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
||||
try:
|
||||
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
||||
state["upgraded_at"] = memory_core.utc_now()
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
try:
|
||||
temporary.write_text(json.dumps(state), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
# Release the claim. The marker still records the old version, so
|
||||
# without this the sentinel makes claim_version_change return early on
|
||||
# every later run and this version's upgrade is never recorded again.
|
||||
for leftover in (sentinel, temporary):
|
||||
try:
|
||||
leftover.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
return previous
|
||||
|
||||
|
||||
def record(
|
||||
@@ -168,19 +492,32 @@ def record(
|
||||
except OSError:
|
||||
pass
|
||||
properties = _safe_value(properties)
|
||||
# Stamped in the RECORDING process, beside harness. `source` used to be
|
||||
# read in the sending process from a module global, so whichever process
|
||||
# drained the spool named every event in it. flush() spreads per-event
|
||||
# properties last, so this now wins over any sender's default.
|
||||
properties.update(
|
||||
harness=_harness,
|
||||
source=_source_tag,
|
||||
plugin_version=memory_core.PLUGIN_VERSION,
|
||||
os=sys.platform,
|
||||
python_version=platform.python_version(),
|
||||
)
|
||||
# Assigned only when the digest is real. _scoped_digest returns "" when
|
||||
# the salt could not be persisted, and an empty property is worse than an
|
||||
# absent one: it survives the None filter below and reads as a value.
|
||||
if repo is not None:
|
||||
properties["repo_hash"] = _digest(getattr(repo, "identity", ""))
|
||||
repo_hash = _scoped_digest(getattr(repo, "identity", ""))
|
||||
if repo_hash:
|
||||
properties["repo_hash"] = repo_hash
|
||||
if session_id:
|
||||
properties["session_hash"] = _digest(session_id)
|
||||
session_hash = _scoped_digest(session_id)
|
||||
if session_hash:
|
||||
properties["session_hash"] = session_hash
|
||||
line = json.dumps(
|
||||
{
|
||||
"event": f"{EVENT_PREFIX}.{event}",
|
||||
"uuid": str(uuid.uuid4()),
|
||||
"timestamp": memory_core.utc_now(),
|
||||
"properties": {
|
||||
key: value for key, value in properties.items() if value is not None
|
||||
@@ -239,38 +576,201 @@ def spawn_flush() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _claim_name(attempt: int = 0) -> str:
|
||||
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
||||
only ever discards a batch that was actually retried and failed."""
|
||||
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
||||
|
||||
|
||||
def _claim_attempt(claim: Path) -> int:
|
||||
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape.
|
||||
|
||||
Anchored on field position, not on a leading "a": the legacy shape is
|
||||
``telemetry-<pid>-<hex>.sending`` and a hex id such as ``a1234567`` would
|
||||
otherwise parse as attempt 1234567 and be discarded unsent on the first
|
||||
flush after an upgrade.
|
||||
"""
|
||||
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
||||
parts = stem.split("-")
|
||||
if len(parts) != 4:
|
||||
return 0
|
||||
tail = parts[3]
|
||||
if tail.startswith("a") and tail[1:].isdigit():
|
||||
return int(tail[1:])
|
||||
return 0
|
||||
|
||||
|
||||
def _touch(path: Path) -> None:
|
||||
"""Refresh mtime so a claim's age measures time since it was claimed.
|
||||
|
||||
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
||||
after a quiet minute inherited the spool's last-write time and looked
|
||||
abandoned the instant it was made. A second sender would then take it over
|
||||
while the first was still posting, and both would deliver the batch.
|
||||
"""
|
||||
try:
|
||||
os.utime(path, None)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _claim_spool() -> Path | None:
|
||||
"""Rename the spool aside so exactly one sender owns each batch."""
|
||||
directory = memory_core.data_dir()
|
||||
claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending"
|
||||
claim = directory / _claim_name()
|
||||
spool = _spool_path()
|
||||
try:
|
||||
spool.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
pass
|
||||
return _claim_parked(directory)
|
||||
|
||||
|
||||
def _sweep_debris(directory: Path) -> None:
|
||||
"""Remove files nothing else will ever pick up again.
|
||||
|
||||
*.partial is a temp file orphaned by a crash between write and rename.
|
||||
*.corrupt is a batch quarantined for undecodable content. No glob in this
|
||||
module matches either, so without this they accumulate on disk for the life
|
||||
of the install.
|
||||
|
||||
Quarantined batches are kept far longer than debris: they are the only
|
||||
evidence left of events that could not be delivered, and someone diagnosing
|
||||
a report of missing telemetry has to be able to find one.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending")):
|
||||
for debris in directory.glob("telemetry-*.partial"):
|
||||
try:
|
||||
if now - debris.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
debris.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
for quarantined in directory.glob("telemetry-*.corrupt"):
|
||||
try:
|
||||
if now - quarantined.stat().st_mtime > CLAIM_EXPIRY_SECONDS:
|
||||
quarantined.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
# The same reasoning covers *.tmp. _write_identity and _install_salt both
|
||||
# create one and unlink it in a finally, which a SIGKILL skips, and no glob
|
||||
# in this module matches the leftovers either.
|
||||
for temporary in directory.glob("telemetry-*.tmp"):
|
||||
try:
|
||||
if now - temporary.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
|
||||
def _claim_parked(directory: Path) -> Path | None:
|
||||
"""Take the oldest abandoned claim, if any lease has actually expired.
|
||||
|
||||
Kept separate from the live spool so flush() can drain both in one run.
|
||||
Previously parked batches were only reachable when no spool existed at all,
|
||||
and because sessions keep recording there usually was one — so a batch
|
||||
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
||||
even though its own presence is what started the sender.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
||||
try:
|
||||
age = now - orphan.stat().st_mtime
|
||||
except OSError:
|
||||
continue
|
||||
if age > CLAIM_EXPIRY_SECONDS:
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
# Someone else holds a live lease on it. This check has to come
|
||||
# first. Claiming a file bumps its attempt count and refreshes its
|
||||
# mtime, so a sender that has just taken the final attempt looks
|
||||
# exhausted to everyone else while it is actively draining. Judging
|
||||
# exhaustion before liveness let a second sender unlink a batch out
|
||||
# from under its owner, losing every event in it.
|
||||
continue
|
||||
# Attempts, not age. Every re-claim touches the mtime and every release
|
||||
# backdates it by a fixed amount, so age is pinned near the stale
|
||||
# threshold and never reaches the expiry. Age stays only as a backstop
|
||||
# for files that never carried an attempt marker.
|
||||
if _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS or age > CLAIM_EXPIRY_SECONDS:
|
||||
try:
|
||||
orphan.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
continue
|
||||
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
||||
try:
|
||||
orphan.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _safe_mtime(path: Path) -> float:
|
||||
try:
|
||||
return path.stat().st_mtime
|
||||
except OSError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
||||
"""Persist the unsent remainder, atomically, and refresh the lease.
|
||||
|
||||
Called after every successful batch. Two jobs: a retry resumes where the
|
||||
send stopped instead of re-posting from the top, and the rewrite doubles as
|
||||
the lease heartbeat, so a slow sender does not have its claim stolen
|
||||
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
||||
"""
|
||||
if not remaining:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return True
|
||||
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
||||
try:
|
||||
payload = "".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining)
|
||||
# fsync before the rename: without it the rename can land while the
|
||||
# bytes have not, and the claim comes back empty or truncated after a
|
||||
# crash. _drain then reads zero events and unlinks it.
|
||||
with open(temporary, "w", encoding="utf-8") as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
temporary.replace(claim)
|
||||
_touch(claim)
|
||||
return True
|
||||
except OSError:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
||||
"""Persist the remainder and drop the lease, because this sender has given up.
|
||||
|
||||
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
||||
make an abandoned batch look actively owned for a further
|
||||
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
||||
threshold lets the next flush pick it up immediately, while the attempt
|
||||
count in the filename still bounds how many times that can happen.
|
||||
"""
|
||||
if not _rewrite_claim(claim, remaining):
|
||||
return
|
||||
try:
|
||||
# Backdate past the stale threshold so the next flush can pick it up,
|
||||
# minus a cooldown that grows with the attempts already spent. Clamped so
|
||||
# the mtime never lands in the future, which would read as a live lease.
|
||||
cooldown = min(_claim_attempt(claim) * RETRY_COOLDOWN_SECONDS, CLAIM_STALE_SECONDS)
|
||||
released = time.time() - CLAIM_STALE_SECONDS - 1 + cooldown
|
||||
os.utime(claim, (released, released))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_email(key: str) -> str:
|
||||
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
||||
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
||||
@@ -300,34 +800,130 @@ def _post(payload: dict[str, Any], url: str) -> bool:
|
||||
|
||||
|
||||
def resolve_distinct_id() -> tuple[str, str]:
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any."""
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
||||
|
||||
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
||||
anonymous id: aliasing one account email to another merges two real person
|
||||
profiles and cannot be undone, so a key that now belongs to a different
|
||||
account re-resolves with no alias.
|
||||
"""
|
||||
identity = _read_identity()
|
||||
email = identity.get("email", "")
|
||||
if email:
|
||||
return email, ""
|
||||
key = memory_core.api_key()
|
||||
fingerprint = _digest(key) if key else ""
|
||||
email = identity.get("email", "")
|
||||
|
||||
if email and fingerprint:
|
||||
recorded = identity.get("key_fingerprint", "")
|
||||
if recorded == fingerprint:
|
||||
return email, ""
|
||||
if not recorded:
|
||||
# Rows written before fingerprints existed. Verify rather than
|
||||
# adopt: a key changed before the upgrade would otherwise bind the
|
||||
# new key to the previous account's email, permanently, and the
|
||||
# fingerprint would then agree with itself forever after.
|
||||
verified = _resolve_email(key)
|
||||
if not verified:
|
||||
# Offline, firewalled, or the API is down. Keep the previous
|
||||
# behaviour and retry on the next flush rather than dropping a
|
||||
# real account attribution. Safe because the same network that
|
||||
# failed /v1/ping/ is about to fail the PostHog POST, so nothing
|
||||
# is delivered under the unverified identity in the meantime.
|
||||
return email, ""
|
||||
identity["email"] = verified
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return verified, ""
|
||||
|
||||
if not key:
|
||||
# No key to verify the account with; do not keep attributing to it.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
email = _resolve_email(key)
|
||||
if not email:
|
||||
|
||||
resolved = _resolve_email(key)
|
||||
if not resolved:
|
||||
# The key changed and will not resolve (revoked, offline, API down).
|
||||
# Reaching here with an email means the recorded fingerprint disagreed,
|
||||
# so the key really did change. Drop the account and rotate: the stored
|
||||
# anonymous id may already be merged into that account's person, and
|
||||
# reusing it would keep the events on the profile we are trying to
|
||||
# leave.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
previous = identity.get("anonymous_id", "")
|
||||
identity["email"] = email
|
||||
|
||||
# Alias only when going anonymous -> email for the first time. Once an anon
|
||||
# id has been merged into an account it must never be offered again: an
|
||||
# alias naming an already-identified id is what could link two real people.
|
||||
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
||||
if previous:
|
||||
identity["aliased"] = True
|
||||
identity["email"] = resolved
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return email, previous
|
||||
return resolved, previous
|
||||
|
||||
|
||||
def flush() -> int:
|
||||
"""Drain claimed spools to PostHog and return the number of events sent."""
|
||||
"""Drain the live spool, then any parked claims, and return events sent."""
|
||||
if not is_enabled():
|
||||
return 0
|
||||
claim = _claim_spool()
|
||||
sent, delivered = _drain(_claim_spool())
|
||||
if not delivered:
|
||||
# The network is failing. Retrying other batches now would only burn
|
||||
# their attempt budget against the same broken connection.
|
||||
return sent
|
||||
|
||||
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
||||
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
||||
directory = memory_core.data_dir()
|
||||
_sweep_debris(directory)
|
||||
for _ in range(MAX_PARKED_PER_RUN):
|
||||
parked = _claim_parked(directory)
|
||||
if parked is None:
|
||||
break
|
||||
count, delivered = _drain(parked)
|
||||
sent += count
|
||||
if not delivered:
|
||||
break
|
||||
return sent
|
||||
|
||||
|
||||
def _drain(claim: Path | None) -> tuple[int, bool]:
|
||||
"""Post one claimed batch file, recording progress after every batch.
|
||||
|
||||
Returns (events sent, whether everything was delivered).
|
||||
"""
|
||||
if claim is None:
|
||||
return 0
|
||||
return 0, True
|
||||
try:
|
||||
lines = claim.read_text(encoding="utf-8").splitlines()
|
||||
except ValueError:
|
||||
# UnicodeDecodeError from a torn write: the content is unrecoverable, so
|
||||
# quarantine rather than retry. flush() runs from a bare `finally:` in
|
||||
# flush_worker, so raising here also skips the handoff cleanup, and an
|
||||
# undecodable file would otherwise be re-read on every flush forever.
|
||||
# Reported as delivered because there is nothing left to deliver and the
|
||||
# rest of the run should continue.
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt"))
|
||||
except OSError:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0, True
|
||||
except OSError:
|
||||
return 0
|
||||
# Could not read it, which is not the same as having nothing to send.
|
||||
# The file is left exactly where it is: a vanished or briefly unreadable
|
||||
# claim is retryable, and quarantining it here would discard events over
|
||||
# a transient filesystem error. Reported as undelivered so the run stops
|
||||
# instead of counting a batch nothing was posted from as delivered.
|
||||
return 0, False
|
||||
events = []
|
||||
for line in lines:
|
||||
try:
|
||||
@@ -337,11 +933,18 @@ def flush() -> int:
|
||||
if isinstance(value, dict) and value.get("event"):
|
||||
events.append(value)
|
||||
if not events:
|
||||
# Only delete when the file really is empty. A non-empty file that
|
||||
# parses to nothing is a torn write, and its contents are the unsent
|
||||
# remainder — deleting it is the data loss this PR exists to prevent.
|
||||
try:
|
||||
claim.unlink()
|
||||
empty = claim.stat().st_size == 0
|
||||
except OSError:
|
||||
empty = True
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt")) if not empty else claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
return 0, True
|
||||
|
||||
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
||||
if aliased_anonymous_id:
|
||||
@@ -360,12 +963,17 @@ def flush() -> int:
|
||||
|
||||
sent = 0
|
||||
for start in range(0, len(events), BATCH_SIZE):
|
||||
chunk = events[start : start + BATCH_SIZE]
|
||||
batch = [
|
||||
{
|
||||
"event": event["event"],
|
||||
"distinct_id": distinct_id,
|
||||
# Carried through from record() so a resend can be collapsed.
|
||||
"uuid": event.get("uuid"),
|
||||
"timestamp": event.get("timestamp"),
|
||||
"properties": {
|
||||
# Fallback only: events recorded by a build before source
|
||||
# moved into record() have none of their own.
|
||||
"source": _source_tag,
|
||||
"language": "python",
|
||||
"$process_person_profile": False,
|
||||
@@ -373,16 +981,24 @@ def flush() -> int:
|
||||
**(event.get("properties") or {}),
|
||||
},
|
||||
}
|
||||
for event in events[start : start + BATCH_SIZE]
|
||||
for event in chunk
|
||||
]
|
||||
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
||||
return sent
|
||||
sent += len(batch)
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return sent
|
||||
# Keep only what has not been delivered, and release the lease.
|
||||
# Previously the whole file was kept and the retry re-posted every
|
||||
# batch, including the ones that had already arrived.
|
||||
_release_claim(claim, events[start:])
|
||||
return sent, False
|
||||
sent += len(chunk)
|
||||
# Record progress and refresh the lease after each successful batch, so
|
||||
# a crash repeats at most one batch instead of the entire file. If the
|
||||
# rewrite fails the claim still holds delivered events, so stop rather
|
||||
# than carry on as though progress were recorded — continuing is how the
|
||||
# duplicate delivery this PR fixes would come back.
|
||||
if not _rewrite_claim(claim, events[start + len(chunk) :]):
|
||||
_release_claim(claim, events[start + len(chunk) :])
|
||||
return sent, False
|
||||
return sent, True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"id": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"homepage": "https://docs.mem0.ai/integrations/codex",
|
||||
"native": {
|
||||
"pluginRoot": "${PLUGIN_ROOT}",
|
||||
|
||||
@@ -7,8 +7,8 @@ disable-model-invocation: true
|
||||
# Pause memory capture
|
||||
|
||||
To pause (hooks stop capturing and sending session content; a minimal
|
||||
anonymous telemetry ping still fires at session start unless
|
||||
`MEM0_TELEMETRY=false`):
|
||||
telemetry ping still fires at session start, under your Mem0 account email,
|
||||
unless `MEM0_TELEMETRY=false`):
|
||||
|
||||
```bash
|
||||
python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "codex" --plugin-data-dir "${PLUGIN_DATA}" pause
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"author": { "name": "Mem0", "email": "support@mem0.ai" },
|
||||
"homepage": "https://docs.mem0.ai/integrations/cursor",
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Generated by integrations/agent-plugin-core/build/build.py. Do not edit."""
|
||||
|
||||
HARNESS_ID = "cursor"
|
||||
SOURCE_TAG = "CURSOR_PLUGIN"
|
||||
|
||||
# Platform-side vocabulary (mem0_event.source + X-Application). The whole
|
||||
# plugin family is one source; which editor it runs in is the application.
|
||||
# An empty application means the host is unknown, and memory_core omits
|
||||
# the header entirely rather than sending a placeholder.
|
||||
PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
PLATFORM_APPLICATION = "cursor"
|
||||
@@ -290,6 +290,11 @@ def run(
|
||||
if args.plugin_data_dir:
|
||||
os.environ[data_dir_env] = args.plugin_data_dir
|
||||
|
||||
# Snapshot BEFORE anything writes to the data dir: cache_plugin_api_key
|
||||
# writes `api-key` and EvidenceStore creates `evidence.sqlite3`, so asking
|
||||
# after them always saw content and every fresh install reported an upgrade.
|
||||
data_dir_was_empty = telemetry.data_dir_was_empty()
|
||||
|
||||
cache_plugin_api_key()
|
||||
if args.action == "session-start":
|
||||
clear_stale_api_key_cache()
|
||||
@@ -305,8 +310,19 @@ def run(
|
||||
return 0
|
||||
|
||||
if args.action == "session-start":
|
||||
if telemetry.is_first_run():
|
||||
# Claims the marker atomically and says which event to record, so a
|
||||
# second session starting alongside this one cannot record it too.
|
||||
first_event = telemetry.claim_install(was_empty=data_dir_was_empty)
|
||||
if first_event == "install":
|
||||
telemetry.record("install")
|
||||
elif first_event == "upgrade":
|
||||
# First run after a build that never wrote the marker; the
|
||||
# predecessor version was never recorded anywhere.
|
||||
telemetry.record("upgrade", from_version="pre-0.3")
|
||||
else:
|
||||
previous = telemetry.claim_version_change()
|
||||
if previous:
|
||||
telemetry.record("upgrade", from_version=previous)
|
||||
recovered = recover_pending_handoffs()
|
||||
record_session_start(store, hook_input)
|
||||
if recovered:
|
||||
|
||||
@@ -29,7 +29,7 @@ from typing import Any, Iterable
|
||||
import telemetry
|
||||
|
||||
DEFAULT_API_URL = "https://api.mem0.ai"
|
||||
PLUGIN_VERSION = "0.3.1"
|
||||
PLUGIN_VERSION = "0.3.2"
|
||||
|
||||
_harness_name: str = "generic"
|
||||
_harness_env_prefix: str = "MEM0_PLUGIN"
|
||||
@@ -1800,6 +1800,34 @@ def extraction_message_batches(
|
||||
return batches
|
||||
|
||||
|
||||
# Platform surface attribution. Read from the generated per-host module so a new
|
||||
# entrypoint is correct without remembering to configure anything.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
except ImportError:
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
|
||||
def platform_headers(key: str) -> dict[str, str]:
|
||||
"""Auth plus the three surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are set-once by contract: this is the
|
||||
outermost layer, so it sets them, and nothing below may overwrite them.
|
||||
X-Mem0-Client is append-only — anything downstream adds itself to the tail.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": _PLATFORM_SOURCE,
|
||||
"X-Mem0-Client": f"mem0-plugin/{PLUGIN_VERSION}",
|
||||
}
|
||||
if _PLATFORM_APPLICATION:
|
||||
headers["X-Application"] = _PLATFORM_APPLICATION
|
||||
return headers
|
||||
|
||||
|
||||
def _request_json(
|
||||
url: str, key: str, payload: dict[str, Any], timeout: float
|
||||
) -> tuple[dict[str, Any] | list[Any], int, int]:
|
||||
@@ -1807,7 +1835,7 @@ def _request_json(
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1834,7 +1862,7 @@ def _get_json(
|
||||
) -> tuple[dict[str, Any] | list[Any], int]:
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="GET",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1980,6 +2008,13 @@ def flush_session(
|
||||
"user_id": write_user,
|
||||
"app_id": repo.app_id,
|
||||
"run_id": session_id,
|
||||
# Top level, not metadata: the backend reads `source` from the body or
|
||||
# the query string, never from metadata, which is where this used to
|
||||
# sit. The X-Mem0-Source header is also read, but only from the
|
||||
# platform release that ships alongside this change, so the body value
|
||||
# is what makes attribution work on both. The harness tag stays in
|
||||
# metadata as hook provenance.
|
||||
"source": _PLATFORM_SOURCE,
|
||||
"metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)},
|
||||
"agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS,
|
||||
"custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS,
|
||||
@@ -2523,7 +2558,7 @@ def _collect_memory_ids(
|
||||
def _delete_memory(api_url: str, key: str, memory_id: str) -> bool:
|
||||
request = urllib.request.Request(
|
||||
f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/",
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Anonymous usage telemetry for Mem0 agent plugins.
|
||||
"""Usage telemetry for Mem0 agent plugins.
|
||||
|
||||
Events are linked to your Mem0 account email when an API key is configured, and
|
||||
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
||||
attribute the same way.
|
||||
|
||||
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
||||
touches the network: `record` appends one JSON line to a local spool and returns.
|
||||
@@ -9,7 +13,8 @@ started once per session and again from the flush worker that is already detache
|
||||
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
||||
|
||||
Never sends prompts, memory text, queries, file paths, repository names, or API
|
||||
keys: only event names, durations, counts, coarse outcomes, and salted hashes.
|
||||
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
||||
identifiers hashed with a random per-install salt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,8 +34,24 @@ from typing import Any
|
||||
|
||||
import memory_core
|
||||
|
||||
_harness: str = "generic"
|
||||
_source_tag: str = "MEM0_PLUGIN"
|
||||
# Seeded from the per-host module the build generates into core/. Two processes
|
||||
# in this pipeline never call init() — mcp_server.py, and the detached
|
||||
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
||||
# was what every one of their events got labelled with.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
||||
except ImportError:
|
||||
_DEFAULT_HARNESS = "generic"
|
||||
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
_salt_cache: str = ""
|
||||
_harness: str = _DEFAULT_HARNESS
|
||||
_source_tag: str = _DEFAULT_SOURCE_TAG
|
||||
_PRIVATE_KEYS = {
|
||||
"apikey",
|
||||
"authorization",
|
||||
@@ -56,10 +77,19 @@ _PRIVATE_KEYS = {
|
||||
}
|
||||
|
||||
|
||||
def init(harness: str = "generic", source_tag: str = "") -> None:
|
||||
def init(harness: str = "", source_tag: str = "") -> None:
|
||||
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
||||
|
||||
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
||||
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
||||
meant one plugin could emit three different source values depending on which
|
||||
process happened to send the batch.
|
||||
"""
|
||||
global _harness, _source_tag
|
||||
_harness = harness
|
||||
_source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN"
|
||||
_harness = harness or _DEFAULT_HARNESS
|
||||
_source_tag = source_tag or (
|
||||
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
||||
)
|
||||
|
||||
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
||||
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
||||
@@ -70,6 +100,16 @@ BATCH_SIZE = 100
|
||||
SEND_TIMEOUT = 5
|
||||
CLAIM_STALE_SECONDS = 120
|
||||
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
||||
# A batch is only discarded once it has genuinely been retried this many times.
|
||||
MAX_CLAIM_ATTEMPTS = 3
|
||||
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
||||
# cannot turn one flush into an unbounded send loop.
|
||||
MAX_PARKED_PER_RUN = 3
|
||||
# Added to the wait before a released claim becomes reclaimable, per attempt
|
||||
# already spent. Releasing straight to "reclaimable now" let two senders burn the
|
||||
# whole budget within seconds of one another on a single momentary failure, and
|
||||
# discard a batch a retry a minute later would have delivered.
|
||||
RETRY_COOLDOWN_SECONDS = 60
|
||||
|
||||
|
||||
def is_enabled() -> bool:
|
||||
@@ -83,9 +123,126 @@ def is_enabled() -> bool:
|
||||
|
||||
|
||||
def _digest(value: str, length: int = 16) -> str:
|
||||
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _salt_path() -> Path:
|
||||
return memory_core.data_dir() / "telemetry-salt"
|
||||
|
||||
|
||||
def _install_salt() -> str:
|
||||
"""Random per-install salt, created once and memoized for the process.
|
||||
|
||||
Deliberately its own file, claimed with O_CREAT|O_EXCL, rather than a key in
|
||||
the identity file. Three reasons, all of which produced wrong data when this
|
||||
lived in the identity dict:
|
||||
|
||||
- Hooks are short-lived separate processes firing on every tool call, and
|
||||
people run more than one agent window. A read-modify-write would let each
|
||||
process mint its own salt, so one repository would hash several ways in the
|
||||
window before a writer won.
|
||||
- resolve_distinct_id holds a copy of the identity dict across a network call
|
||||
to /v1/ping/, so whichever write landed second erased the other's key —
|
||||
losing either the salt (repo_hash changes mid-stream) or the email (a
|
||||
second $identify, splitting the person).
|
||||
- Touching the identity file from record() would create it, and is_first_run
|
||||
keys off that file, so recording an event would silently suppress the
|
||||
install event.
|
||||
|
||||
Published atomically, and there is deliberately no derived fallback. Creating
|
||||
the file with O_CREAT|O_EXCL and then writing into it leaves a window where
|
||||
the file exists and is empty, and a concurrent hook that reads it in that
|
||||
window gets nothing. Falling back to a digest of the path would hand that
|
||||
process a salt an attacker can compute, memoized for its whole run, which is
|
||||
the privacy control this function exists to provide silently turning itself
|
||||
off under load. The salt is written to a private temp file first and linked
|
||||
into place, so the name either does not exist or already has the full value.
|
||||
|
||||
Returns "" when it genuinely cannot persist. Callers omit the hash entirely
|
||||
rather than emit an unsalted one.
|
||||
"""
|
||||
global _salt_cache
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
|
||||
path = _salt_path()
|
||||
# Read before writing. Hooks are separate processes firing on every tool
|
||||
# call, so all but the first find the salt already published; going straight
|
||||
# to create-fsync-link-unlink meant every one of them paid an fsync to
|
||||
# discover that, on a path whose whole promise is appending a line and
|
||||
# returning.
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
temporary = path.with_name(f"{path.name}.{os.getpid()}.tmp")
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
stream.write(uuid.uuid4().hex)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
try:
|
||||
# Atomic claim: fails if another process already published one.
|
||||
# os.link rather than replace, which would clobber theirs.
|
||||
os.link(temporary, path)
|
||||
except FileExistsError:
|
||||
pass
|
||||
except OSError:
|
||||
# No hardlinks here (some network mounts, some container volumes).
|
||||
# Claim the name directly instead. That reopens the empty-file
|
||||
# window, but the window is now benign: a reader that lands in it
|
||||
# gets "" and omits the hash for that process rather than caching a
|
||||
# guessable one. Losing the hashes on every run of an entire
|
||||
# filesystem is the worse failure.
|
||||
try:
|
||||
fallback = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fallback, "w", encoding="utf-8") as stream:
|
||||
stream.write(temporary.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_salt_cache = ""
|
||||
return _salt_cache
|
||||
|
||||
|
||||
def _scoped_digest(value: str, length: int = 16) -> str:
|
||||
"""Salted digest for values drawn from a guessable space.
|
||||
|
||||
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
||||
no remote — which normally contains the account username. Sixteen unsalted
|
||||
hex characters over that input space is enumerable, so this is not a
|
||||
privacy control without the salt. Salting per install keeps every
|
||||
within-account join the analytics actually use and gives up only
|
||||
cross-machine joins on the same repository, which nothing computes.
|
||||
|
||||
Returns "" when there is no salt, so record() omits the property. An
|
||||
unsalted digest over this input space is close to plaintext, and emitting one
|
||||
under a name that implies it is hashed is worse than sending nothing.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
salt = _install_salt()
|
||||
if not salt:
|
||||
return ""
|
||||
return hashlib.sha256(f"{salt}:{value}".encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _safe_value(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return memory_core.redact(value)
|
||||
@@ -145,9 +302,176 @@ def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
||||
return created
|
||||
|
||||
|
||||
def _rotate_anonymous_id(identity: dict[str, str]) -> str:
|
||||
"""Mint a fresh anonymous id because the account context is gone.
|
||||
|
||||
The previous id may already have been merged into a person profile by an
|
||||
$identify, and that merge is permanent. Reusing it after a logout or a key
|
||||
change attributes everything that follows to the account that just went
|
||||
away, which is the same misattribution the key fingerprint exists to stop,
|
||||
only arriving through the anonymous path instead.
|
||||
|
||||
`aliased` is cleared with it: the new id has never been merged, so it is
|
||||
eligible to be aliased into whatever account comes next.
|
||||
"""
|
||||
created = f"code-anon-{uuid.uuid4().hex}"
|
||||
identity["anonymous_id"] = created
|
||||
identity.pop("aliased", None)
|
||||
_write_identity(identity)
|
||||
return created
|
||||
|
||||
|
||||
def _install_state_path() -> Path:
|
||||
return memory_core.data_dir() / "install-state.json"
|
||||
|
||||
|
||||
def is_first_run() -> bool:
|
||||
"""Whether this machine has never recorded a plugin event before."""
|
||||
return not _identity_path().exists()
|
||||
"""Whether install has never been recorded on this machine.
|
||||
|
||||
Deliberately NOT the identity file. That file is only written by a
|
||||
successful flush, so an offline or firewalled user recorded code.install on
|
||||
every single session, forever — and every 0.2.x user recorded one on their
|
||||
first 0.3.x session because 0.2.x never wrote it at all.
|
||||
"""
|
||||
return not _install_state_path().exists()
|
||||
|
||||
|
||||
def data_dir_was_empty() -> bool:
|
||||
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
||||
|
||||
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
||||
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
||||
asking at claim time always saw content and every fresh install reported an
|
||||
upgrade. The caller snapshots this at the top of the run instead.
|
||||
"""
|
||||
return not _data_dir_has_content()
|
||||
|
||||
|
||||
def claim_install(was_empty: bool | None = None) -> str | None:
|
||||
"""Claim the one install/upgrade record for this machine, atomically.
|
||||
|
||||
Returns the event to record ("install" or "upgrade"), or None if another
|
||||
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
||||
cannot both win.
|
||||
|
||||
`was_empty` must come from data_dir_was_empty() called before this process
|
||||
wrote anything. Omitting it falls back to checking now, which is only
|
||||
correct for a caller that has touched nothing.
|
||||
"""
|
||||
if not is_enabled():
|
||||
# Never consume the one-shot claim while the user is opted out, or they
|
||||
# would silently lose their install event if they later opt in.
|
||||
return None
|
||||
|
||||
path = _install_state_path()
|
||||
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
json.dump(
|
||||
{
|
||||
"plugin_version": memory_core.PLUGIN_VERSION,
|
||||
"installed_at": memory_core.utc_now(),
|
||||
"upgraded": upgrading,
|
||||
},
|
||||
stream,
|
||||
)
|
||||
# Durable before this returns. The O_EXCL open is what makes the
|
||||
# claim exclusive, so it cannot be replaced by a temp-and-rename
|
||||
# without losing that, which leaves the content as the thing to make
|
||||
# safe. A kill between the open and this fsync used to leave a marker
|
||||
# that exists but parses to nothing: is_first_run reads it as claimed
|
||||
# and claim_version_change cannot read a version out of it.
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
except OSError:
|
||||
pass
|
||||
return "upgrade" if upgrading else "install"
|
||||
|
||||
|
||||
def _data_dir_has_content() -> bool:
|
||||
"""Whether anything predates this session in the plugin data directory."""
|
||||
try:
|
||||
for entry in memory_core.data_dir().iterdir():
|
||||
if entry.name != "install-state.json":
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _repair_install_state(path: Path) -> None:
|
||||
"""Rewrite an unparseable marker so version tracking can resume."""
|
||||
try:
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
temporary.write_text(
|
||||
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def claim_version_change() -> str | None:
|
||||
"""Return the previously recorded version if it differs, updating the marker.
|
||||
|
||||
Only meaningful once the marker exists — the first transition into 0.3.x has
|
||||
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
||||
the marker means the next session sees no change and records nothing.
|
||||
"""
|
||||
path = _install_state_path()
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
return None
|
||||
except json.JSONDecodeError:
|
||||
# A crash between O_EXCL and the write leaves an empty marker. Left
|
||||
# alone it disables every future upgrade event on this machine, because
|
||||
# claim_install sees the file and this function cannot parse it.
|
||||
state = None
|
||||
if not isinstance(state, dict):
|
||||
_repair_install_state(path)
|
||||
return None
|
||||
previous = str(state.get("plugin_version") or "")
|
||||
if not previous or previous == memory_core.PLUGIN_VERSION:
|
||||
return None
|
||||
# Claim the transition with an exclusive sentinel before rewriting the
|
||||
# marker. A plain read-modify-write let every concurrently starting session
|
||||
# observe the old version and each record its own upgrade — and the first
|
||||
# session after a version bump is exactly when several agent windows restart
|
||||
# together.
|
||||
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
||||
try:
|
||||
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
||||
state["upgraded_at"] = memory_core.utc_now()
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
try:
|
||||
temporary.write_text(json.dumps(state), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
# Release the claim. The marker still records the old version, so
|
||||
# without this the sentinel makes claim_version_change return early on
|
||||
# every later run and this version's upgrade is never recorded again.
|
||||
for leftover in (sentinel, temporary):
|
||||
try:
|
||||
leftover.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
return previous
|
||||
|
||||
|
||||
def record(
|
||||
@@ -168,19 +492,32 @@ def record(
|
||||
except OSError:
|
||||
pass
|
||||
properties = _safe_value(properties)
|
||||
# Stamped in the RECORDING process, beside harness. `source` used to be
|
||||
# read in the sending process from a module global, so whichever process
|
||||
# drained the spool named every event in it. flush() spreads per-event
|
||||
# properties last, so this now wins over any sender's default.
|
||||
properties.update(
|
||||
harness=_harness,
|
||||
source=_source_tag,
|
||||
plugin_version=memory_core.PLUGIN_VERSION,
|
||||
os=sys.platform,
|
||||
python_version=platform.python_version(),
|
||||
)
|
||||
# Assigned only when the digest is real. _scoped_digest returns "" when
|
||||
# the salt could not be persisted, and an empty property is worse than an
|
||||
# absent one: it survives the None filter below and reads as a value.
|
||||
if repo is not None:
|
||||
properties["repo_hash"] = _digest(getattr(repo, "identity", ""))
|
||||
repo_hash = _scoped_digest(getattr(repo, "identity", ""))
|
||||
if repo_hash:
|
||||
properties["repo_hash"] = repo_hash
|
||||
if session_id:
|
||||
properties["session_hash"] = _digest(session_id)
|
||||
session_hash = _scoped_digest(session_id)
|
||||
if session_hash:
|
||||
properties["session_hash"] = session_hash
|
||||
line = json.dumps(
|
||||
{
|
||||
"event": f"{EVENT_PREFIX}.{event}",
|
||||
"uuid": str(uuid.uuid4()),
|
||||
"timestamp": memory_core.utc_now(),
|
||||
"properties": {
|
||||
key: value for key, value in properties.items() if value is not None
|
||||
@@ -239,38 +576,201 @@ def spawn_flush() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _claim_name(attempt: int = 0) -> str:
|
||||
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
||||
only ever discards a batch that was actually retried and failed."""
|
||||
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
||||
|
||||
|
||||
def _claim_attempt(claim: Path) -> int:
|
||||
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape.
|
||||
|
||||
Anchored on field position, not on a leading "a": the legacy shape is
|
||||
``telemetry-<pid>-<hex>.sending`` and a hex id such as ``a1234567`` would
|
||||
otherwise parse as attempt 1234567 and be discarded unsent on the first
|
||||
flush after an upgrade.
|
||||
"""
|
||||
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
||||
parts = stem.split("-")
|
||||
if len(parts) != 4:
|
||||
return 0
|
||||
tail = parts[3]
|
||||
if tail.startswith("a") and tail[1:].isdigit():
|
||||
return int(tail[1:])
|
||||
return 0
|
||||
|
||||
|
||||
def _touch(path: Path) -> None:
|
||||
"""Refresh mtime so a claim's age measures time since it was claimed.
|
||||
|
||||
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
||||
after a quiet minute inherited the spool's last-write time and looked
|
||||
abandoned the instant it was made. A second sender would then take it over
|
||||
while the first was still posting, and both would deliver the batch.
|
||||
"""
|
||||
try:
|
||||
os.utime(path, None)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _claim_spool() -> Path | None:
|
||||
"""Rename the spool aside so exactly one sender owns each batch."""
|
||||
directory = memory_core.data_dir()
|
||||
claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending"
|
||||
claim = directory / _claim_name()
|
||||
spool = _spool_path()
|
||||
try:
|
||||
spool.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
pass
|
||||
return _claim_parked(directory)
|
||||
|
||||
|
||||
def _sweep_debris(directory: Path) -> None:
|
||||
"""Remove files nothing else will ever pick up again.
|
||||
|
||||
*.partial is a temp file orphaned by a crash between write and rename.
|
||||
*.corrupt is a batch quarantined for undecodable content. No glob in this
|
||||
module matches either, so without this they accumulate on disk for the life
|
||||
of the install.
|
||||
|
||||
Quarantined batches are kept far longer than debris: they are the only
|
||||
evidence left of events that could not be delivered, and someone diagnosing
|
||||
a report of missing telemetry has to be able to find one.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending")):
|
||||
for debris in directory.glob("telemetry-*.partial"):
|
||||
try:
|
||||
if now - debris.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
debris.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
for quarantined in directory.glob("telemetry-*.corrupt"):
|
||||
try:
|
||||
if now - quarantined.stat().st_mtime > CLAIM_EXPIRY_SECONDS:
|
||||
quarantined.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
# The same reasoning covers *.tmp. _write_identity and _install_salt both
|
||||
# create one and unlink it in a finally, which a SIGKILL skips, and no glob
|
||||
# in this module matches the leftovers either.
|
||||
for temporary in directory.glob("telemetry-*.tmp"):
|
||||
try:
|
||||
if now - temporary.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
|
||||
def _claim_parked(directory: Path) -> Path | None:
|
||||
"""Take the oldest abandoned claim, if any lease has actually expired.
|
||||
|
||||
Kept separate from the live spool so flush() can drain both in one run.
|
||||
Previously parked batches were only reachable when no spool existed at all,
|
||||
and because sessions keep recording there usually was one — so a batch
|
||||
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
||||
even though its own presence is what started the sender.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
||||
try:
|
||||
age = now - orphan.stat().st_mtime
|
||||
except OSError:
|
||||
continue
|
||||
if age > CLAIM_EXPIRY_SECONDS:
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
# Someone else holds a live lease on it. This check has to come
|
||||
# first. Claiming a file bumps its attempt count and refreshes its
|
||||
# mtime, so a sender that has just taken the final attempt looks
|
||||
# exhausted to everyone else while it is actively draining. Judging
|
||||
# exhaustion before liveness let a second sender unlink a batch out
|
||||
# from under its owner, losing every event in it.
|
||||
continue
|
||||
# Attempts, not age. Every re-claim touches the mtime and every release
|
||||
# backdates it by a fixed amount, so age is pinned near the stale
|
||||
# threshold and never reaches the expiry. Age stays only as a backstop
|
||||
# for files that never carried an attempt marker.
|
||||
if _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS or age > CLAIM_EXPIRY_SECONDS:
|
||||
try:
|
||||
orphan.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
continue
|
||||
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
||||
try:
|
||||
orphan.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _safe_mtime(path: Path) -> float:
|
||||
try:
|
||||
return path.stat().st_mtime
|
||||
except OSError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
||||
"""Persist the unsent remainder, atomically, and refresh the lease.
|
||||
|
||||
Called after every successful batch. Two jobs: a retry resumes where the
|
||||
send stopped instead of re-posting from the top, and the rewrite doubles as
|
||||
the lease heartbeat, so a slow sender does not have its claim stolen
|
||||
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
||||
"""
|
||||
if not remaining:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return True
|
||||
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
||||
try:
|
||||
payload = "".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining)
|
||||
# fsync before the rename: without it the rename can land while the
|
||||
# bytes have not, and the claim comes back empty or truncated after a
|
||||
# crash. _drain then reads zero events and unlinks it.
|
||||
with open(temporary, "w", encoding="utf-8") as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
temporary.replace(claim)
|
||||
_touch(claim)
|
||||
return True
|
||||
except OSError:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
||||
"""Persist the remainder and drop the lease, because this sender has given up.
|
||||
|
||||
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
||||
make an abandoned batch look actively owned for a further
|
||||
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
||||
threshold lets the next flush pick it up immediately, while the attempt
|
||||
count in the filename still bounds how many times that can happen.
|
||||
"""
|
||||
if not _rewrite_claim(claim, remaining):
|
||||
return
|
||||
try:
|
||||
# Backdate past the stale threshold so the next flush can pick it up,
|
||||
# minus a cooldown that grows with the attempts already spent. Clamped so
|
||||
# the mtime never lands in the future, which would read as a live lease.
|
||||
cooldown = min(_claim_attempt(claim) * RETRY_COOLDOWN_SECONDS, CLAIM_STALE_SECONDS)
|
||||
released = time.time() - CLAIM_STALE_SECONDS - 1 + cooldown
|
||||
os.utime(claim, (released, released))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_email(key: str) -> str:
|
||||
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
||||
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
||||
@@ -300,34 +800,130 @@ def _post(payload: dict[str, Any], url: str) -> bool:
|
||||
|
||||
|
||||
def resolve_distinct_id() -> tuple[str, str]:
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any."""
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
||||
|
||||
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
||||
anonymous id: aliasing one account email to another merges two real person
|
||||
profiles and cannot be undone, so a key that now belongs to a different
|
||||
account re-resolves with no alias.
|
||||
"""
|
||||
identity = _read_identity()
|
||||
email = identity.get("email", "")
|
||||
if email:
|
||||
return email, ""
|
||||
key = memory_core.api_key()
|
||||
fingerprint = _digest(key) if key else ""
|
||||
email = identity.get("email", "")
|
||||
|
||||
if email and fingerprint:
|
||||
recorded = identity.get("key_fingerprint", "")
|
||||
if recorded == fingerprint:
|
||||
return email, ""
|
||||
if not recorded:
|
||||
# Rows written before fingerprints existed. Verify rather than
|
||||
# adopt: a key changed before the upgrade would otherwise bind the
|
||||
# new key to the previous account's email, permanently, and the
|
||||
# fingerprint would then agree with itself forever after.
|
||||
verified = _resolve_email(key)
|
||||
if not verified:
|
||||
# Offline, firewalled, or the API is down. Keep the previous
|
||||
# behaviour and retry on the next flush rather than dropping a
|
||||
# real account attribution. Safe because the same network that
|
||||
# failed /v1/ping/ is about to fail the PostHog POST, so nothing
|
||||
# is delivered under the unverified identity in the meantime.
|
||||
return email, ""
|
||||
identity["email"] = verified
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return verified, ""
|
||||
|
||||
if not key:
|
||||
# No key to verify the account with; do not keep attributing to it.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
email = _resolve_email(key)
|
||||
if not email:
|
||||
|
||||
resolved = _resolve_email(key)
|
||||
if not resolved:
|
||||
# The key changed and will not resolve (revoked, offline, API down).
|
||||
# Reaching here with an email means the recorded fingerprint disagreed,
|
||||
# so the key really did change. Drop the account and rotate: the stored
|
||||
# anonymous id may already be merged into that account's person, and
|
||||
# reusing it would keep the events on the profile we are trying to
|
||||
# leave.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
previous = identity.get("anonymous_id", "")
|
||||
identity["email"] = email
|
||||
|
||||
# Alias only when going anonymous -> email for the first time. Once an anon
|
||||
# id has been merged into an account it must never be offered again: an
|
||||
# alias naming an already-identified id is what could link two real people.
|
||||
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
||||
if previous:
|
||||
identity["aliased"] = True
|
||||
identity["email"] = resolved
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return email, previous
|
||||
return resolved, previous
|
||||
|
||||
|
||||
def flush() -> int:
|
||||
"""Drain claimed spools to PostHog and return the number of events sent."""
|
||||
"""Drain the live spool, then any parked claims, and return events sent."""
|
||||
if not is_enabled():
|
||||
return 0
|
||||
claim = _claim_spool()
|
||||
sent, delivered = _drain(_claim_spool())
|
||||
if not delivered:
|
||||
# The network is failing. Retrying other batches now would only burn
|
||||
# their attempt budget against the same broken connection.
|
||||
return sent
|
||||
|
||||
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
||||
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
||||
directory = memory_core.data_dir()
|
||||
_sweep_debris(directory)
|
||||
for _ in range(MAX_PARKED_PER_RUN):
|
||||
parked = _claim_parked(directory)
|
||||
if parked is None:
|
||||
break
|
||||
count, delivered = _drain(parked)
|
||||
sent += count
|
||||
if not delivered:
|
||||
break
|
||||
return sent
|
||||
|
||||
|
||||
def _drain(claim: Path | None) -> tuple[int, bool]:
|
||||
"""Post one claimed batch file, recording progress after every batch.
|
||||
|
||||
Returns (events sent, whether everything was delivered).
|
||||
"""
|
||||
if claim is None:
|
||||
return 0
|
||||
return 0, True
|
||||
try:
|
||||
lines = claim.read_text(encoding="utf-8").splitlines()
|
||||
except ValueError:
|
||||
# UnicodeDecodeError from a torn write: the content is unrecoverable, so
|
||||
# quarantine rather than retry. flush() runs from a bare `finally:` in
|
||||
# flush_worker, so raising here also skips the handoff cleanup, and an
|
||||
# undecodable file would otherwise be re-read on every flush forever.
|
||||
# Reported as delivered because there is nothing left to deliver and the
|
||||
# rest of the run should continue.
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt"))
|
||||
except OSError:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0, True
|
||||
except OSError:
|
||||
return 0
|
||||
# Could not read it, which is not the same as having nothing to send.
|
||||
# The file is left exactly where it is: a vanished or briefly unreadable
|
||||
# claim is retryable, and quarantining it here would discard events over
|
||||
# a transient filesystem error. Reported as undelivered so the run stops
|
||||
# instead of counting a batch nothing was posted from as delivered.
|
||||
return 0, False
|
||||
events = []
|
||||
for line in lines:
|
||||
try:
|
||||
@@ -337,11 +933,18 @@ def flush() -> int:
|
||||
if isinstance(value, dict) and value.get("event"):
|
||||
events.append(value)
|
||||
if not events:
|
||||
# Only delete when the file really is empty. A non-empty file that
|
||||
# parses to nothing is a torn write, and its contents are the unsent
|
||||
# remainder — deleting it is the data loss this PR exists to prevent.
|
||||
try:
|
||||
claim.unlink()
|
||||
empty = claim.stat().st_size == 0
|
||||
except OSError:
|
||||
empty = True
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt")) if not empty else claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
return 0, True
|
||||
|
||||
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
||||
if aliased_anonymous_id:
|
||||
@@ -360,12 +963,17 @@ def flush() -> int:
|
||||
|
||||
sent = 0
|
||||
for start in range(0, len(events), BATCH_SIZE):
|
||||
chunk = events[start : start + BATCH_SIZE]
|
||||
batch = [
|
||||
{
|
||||
"event": event["event"],
|
||||
"distinct_id": distinct_id,
|
||||
# Carried through from record() so a resend can be collapsed.
|
||||
"uuid": event.get("uuid"),
|
||||
"timestamp": event.get("timestamp"),
|
||||
"properties": {
|
||||
# Fallback only: events recorded by a build before source
|
||||
# moved into record() have none of their own.
|
||||
"source": _source_tag,
|
||||
"language": "python",
|
||||
"$process_person_profile": False,
|
||||
@@ -373,16 +981,24 @@ def flush() -> int:
|
||||
**(event.get("properties") or {}),
|
||||
},
|
||||
}
|
||||
for event in events[start : start + BATCH_SIZE]
|
||||
for event in chunk
|
||||
]
|
||||
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
||||
return sent
|
||||
sent += len(batch)
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return sent
|
||||
# Keep only what has not been delivered, and release the lease.
|
||||
# Previously the whole file was kept and the retry re-posted every
|
||||
# batch, including the ones that had already arrived.
|
||||
_release_claim(claim, events[start:])
|
||||
return sent, False
|
||||
sent += len(chunk)
|
||||
# Record progress and refresh the lease after each successful batch, so
|
||||
# a crash repeats at most one batch instead of the entire file. If the
|
||||
# rewrite fails the claim still holds delivered events, so stop rather
|
||||
# than carry on as though progress were recorded — continuing is how the
|
||||
# duplicate delivery this PR fixes would come back.
|
||||
if not _rewrite_claim(claim, events[start + len(chunk) :]):
|
||||
_release_claim(claim, events[start + len(chunk) :])
|
||||
return sent, False
|
||||
return sent, True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"id": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"homepage": "https://docs.mem0.ai/integrations/cursor",
|
||||
"native": {
|
||||
"pluginRoot": "${CURSOR_PLUGIN_ROOT}",
|
||||
|
||||
@@ -7,8 +7,8 @@ disable-model-invocation: true
|
||||
# Pause memory capture
|
||||
|
||||
To pause (hooks stop capturing and sending session content; a minimal
|
||||
anonymous telemetry ping still fires at session start unless
|
||||
`MEM0_TELEMETRY=false`):
|
||||
telemetry ping still fires at session start, under your Mem0 account email,
|
||||
unless `MEM0_TELEMETRY=false`):
|
||||
|
||||
```bash
|
||||
python3 "${CURSOR_PLUGIN_ROOT}/core/memory_cli.py" --harness "cursor" pause
|
||||
|
||||
@@ -86,9 +86,9 @@ Per-call `userId` overrides are rejected unless the operator enables `allowUserO
|
||||
|
||||
## Telemetry
|
||||
|
||||
Writes are tagged `source="DEEPSEEK_HARNESS"` so Mem0's backend can attribute usage to this integration. For it to surface by name (rather than bucketing into `OTHERS`), `DEEPSEEK_HARNESS` must be present in the backend's `KNOWN_EVENT_SOURCES` allowlist, a one-line platform change matching the existing `ZAPIER` / `STRANDS` sources.
|
||||
Writes are tagged `source="DEEPSEEK_HARNESS"`. That value has to exist in the backend's `EventSource` enum for usage to surface by name; until it does, these writes read as `OTHERS`. It is added by [mem0ai/platform#3602](https://github.com/mem0ai/platform/pull/3602), which has to ship before this claim is true.
|
||||
|
||||
The plugin also sends anonymous usage events (which tool ran, duration, result counts, coarse failure kind) so Mem0 can tell how the plugin is used and where it breaks. Queries, memory text, and entity ids are never sent. Turn it off with `MEM0_TELEMETRY=false`.
|
||||
The plugin also sends usage events (which tool ran, duration, result counts, coarse failure kind) so Mem0 can tell how the plugin is used and where it breaks. These are **not anonymous**: when an API key is configured they are sent under your Mem0 account email, the same way the SDK attributes its own. Queries, memory text, and entity ids are never sent. Turn it off with `MEM0_TELEMETRY=false`.
|
||||
|
||||
## Status
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@mem0/deepseek-plugin",
|
||||
"version": "0.3.0",
|
||||
"version": "0.3.1",
|
||||
"description": "Mem0 long-term memory as a native DeepSeek Harness (Cordis) plugin.",
|
||||
"type": "module",
|
||||
"license": "Apache-2.0",
|
||||
|
||||
@@ -26,10 +26,9 @@ export const name = "mem0";
|
||||
export const inject = ["tools", "systemPrompt"];
|
||||
|
||||
// Tags writes so Mem0's backend attributes them to this integration in
|
||||
// telemetry. The backend keeps recognized values via its KNOWN_EVENT_SOURCES
|
||||
// allowlist; unknown values bucket into "OTHERS", so "DEEPSEEK_HARNESS" must be
|
||||
// added to that allowlist for usage to surface by name (a one-line backend PR,
|
||||
// same pattern as the ZAPIER / STRANDS sources).
|
||||
// telemetry. Values outside the backend's KNOWN_EVENT_SOURCES allowlist bucket
|
||||
// into "OTHERS"; this one is added by mem0ai/platform#3602 and reads as OTHERS
|
||||
// until that ships.
|
||||
const SOURCE = "DEEPSEEK_HARNESS";
|
||||
|
||||
const DEFAULT_SEARCH_LIMIT = 10;
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Generated by integrations/agent-plugin-core/build/build.py. Do not edit."""
|
||||
|
||||
HARNESS_ID = "kimi"
|
||||
SOURCE_TAG = "KIMI_PLUGIN"
|
||||
|
||||
# Platform-side vocabulary (mem0_event.source + X-Application). The whole
|
||||
# plugin family is one source; which editor it runs in is the application.
|
||||
# An empty application means the host is unknown, and memory_core omits
|
||||
# the header entirely rather than sending a placeholder.
|
||||
PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
PLATFORM_APPLICATION = "kimi"
|
||||
@@ -290,6 +290,11 @@ def run(
|
||||
if args.plugin_data_dir:
|
||||
os.environ[data_dir_env] = args.plugin_data_dir
|
||||
|
||||
# Snapshot BEFORE anything writes to the data dir: cache_plugin_api_key
|
||||
# writes `api-key` and EvidenceStore creates `evidence.sqlite3`, so asking
|
||||
# after them always saw content and every fresh install reported an upgrade.
|
||||
data_dir_was_empty = telemetry.data_dir_was_empty()
|
||||
|
||||
cache_plugin_api_key()
|
||||
if args.action == "session-start":
|
||||
clear_stale_api_key_cache()
|
||||
@@ -305,8 +310,19 @@ def run(
|
||||
return 0
|
||||
|
||||
if args.action == "session-start":
|
||||
if telemetry.is_first_run():
|
||||
# Claims the marker atomically and says which event to record, so a
|
||||
# second session starting alongside this one cannot record it too.
|
||||
first_event = telemetry.claim_install(was_empty=data_dir_was_empty)
|
||||
if first_event == "install":
|
||||
telemetry.record("install")
|
||||
elif first_event == "upgrade":
|
||||
# First run after a build that never wrote the marker; the
|
||||
# predecessor version was never recorded anywhere.
|
||||
telemetry.record("upgrade", from_version="pre-0.3")
|
||||
else:
|
||||
previous = telemetry.claim_version_change()
|
||||
if previous:
|
||||
telemetry.record("upgrade", from_version=previous)
|
||||
recovered = recover_pending_handoffs()
|
||||
record_session_start(store, hook_input)
|
||||
if recovered:
|
||||
|
||||
@@ -29,7 +29,7 @@ from typing import Any, Iterable
|
||||
import telemetry
|
||||
|
||||
DEFAULT_API_URL = "https://api.mem0.ai"
|
||||
PLUGIN_VERSION = "0.3.1"
|
||||
PLUGIN_VERSION = "0.3.2"
|
||||
|
||||
_harness_name: str = "generic"
|
||||
_harness_env_prefix: str = "MEM0_PLUGIN"
|
||||
@@ -1800,6 +1800,34 @@ def extraction_message_batches(
|
||||
return batches
|
||||
|
||||
|
||||
# Platform surface attribution. Read from the generated per-host module so a new
|
||||
# entrypoint is correct without remembering to configure anything.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
except ImportError:
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
|
||||
def platform_headers(key: str) -> dict[str, str]:
|
||||
"""Auth plus the three surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are set-once by contract: this is the
|
||||
outermost layer, so it sets them, and nothing below may overwrite them.
|
||||
X-Mem0-Client is append-only — anything downstream adds itself to the tail.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": _PLATFORM_SOURCE,
|
||||
"X-Mem0-Client": f"mem0-plugin/{PLUGIN_VERSION}",
|
||||
}
|
||||
if _PLATFORM_APPLICATION:
|
||||
headers["X-Application"] = _PLATFORM_APPLICATION
|
||||
return headers
|
||||
|
||||
|
||||
def _request_json(
|
||||
url: str, key: str, payload: dict[str, Any], timeout: float
|
||||
) -> tuple[dict[str, Any] | list[Any], int, int]:
|
||||
@@ -1807,7 +1835,7 @@ def _request_json(
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1834,7 +1862,7 @@ def _get_json(
|
||||
) -> tuple[dict[str, Any] | list[Any], int]:
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="GET",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1980,6 +2008,13 @@ def flush_session(
|
||||
"user_id": write_user,
|
||||
"app_id": repo.app_id,
|
||||
"run_id": session_id,
|
||||
# Top level, not metadata: the backend reads `source` from the body or
|
||||
# the query string, never from metadata, which is where this used to
|
||||
# sit. The X-Mem0-Source header is also read, but only from the
|
||||
# platform release that ships alongside this change, so the body value
|
||||
# is what makes attribution work on both. The harness tag stays in
|
||||
# metadata as hook provenance.
|
||||
"source": _PLATFORM_SOURCE,
|
||||
"metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)},
|
||||
"agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS,
|
||||
"custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS,
|
||||
@@ -2523,7 +2558,7 @@ def _collect_memory_ids(
|
||||
def _delete_memory(api_url: str, key: str, memory_id: str) -> bool:
|
||||
request = urllib.request.Request(
|
||||
f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/",
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Anonymous usage telemetry for Mem0 agent plugins.
|
||||
"""Usage telemetry for Mem0 agent plugins.
|
||||
|
||||
Events are linked to your Mem0 account email when an API key is configured, and
|
||||
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
||||
attribute the same way.
|
||||
|
||||
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
||||
touches the network: `record` appends one JSON line to a local spool and returns.
|
||||
@@ -9,7 +13,8 @@ started once per session and again from the flush worker that is already detache
|
||||
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
||||
|
||||
Never sends prompts, memory text, queries, file paths, repository names, or API
|
||||
keys: only event names, durations, counts, coarse outcomes, and salted hashes.
|
||||
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
||||
identifiers hashed with a random per-install salt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,8 +34,24 @@ from typing import Any
|
||||
|
||||
import memory_core
|
||||
|
||||
_harness: str = "generic"
|
||||
_source_tag: str = "MEM0_PLUGIN"
|
||||
# Seeded from the per-host module the build generates into core/. Two processes
|
||||
# in this pipeline never call init() — mcp_server.py, and the detached
|
||||
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
||||
# was what every one of their events got labelled with.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
||||
except ImportError:
|
||||
_DEFAULT_HARNESS = "generic"
|
||||
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
_salt_cache: str = ""
|
||||
_harness: str = _DEFAULT_HARNESS
|
||||
_source_tag: str = _DEFAULT_SOURCE_TAG
|
||||
_PRIVATE_KEYS = {
|
||||
"apikey",
|
||||
"authorization",
|
||||
@@ -56,10 +77,19 @@ _PRIVATE_KEYS = {
|
||||
}
|
||||
|
||||
|
||||
def init(harness: str = "generic", source_tag: str = "") -> None:
|
||||
def init(harness: str = "", source_tag: str = "") -> None:
|
||||
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
||||
|
||||
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
||||
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
||||
meant one plugin could emit three different source values depending on which
|
||||
process happened to send the batch.
|
||||
"""
|
||||
global _harness, _source_tag
|
||||
_harness = harness
|
||||
_source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN"
|
||||
_harness = harness or _DEFAULT_HARNESS
|
||||
_source_tag = source_tag or (
|
||||
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
||||
)
|
||||
|
||||
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
||||
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
||||
@@ -70,6 +100,16 @@ BATCH_SIZE = 100
|
||||
SEND_TIMEOUT = 5
|
||||
CLAIM_STALE_SECONDS = 120
|
||||
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
||||
# A batch is only discarded once it has genuinely been retried this many times.
|
||||
MAX_CLAIM_ATTEMPTS = 3
|
||||
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
||||
# cannot turn one flush into an unbounded send loop.
|
||||
MAX_PARKED_PER_RUN = 3
|
||||
# Added to the wait before a released claim becomes reclaimable, per attempt
|
||||
# already spent. Releasing straight to "reclaimable now" let two senders burn the
|
||||
# whole budget within seconds of one another on a single momentary failure, and
|
||||
# discard a batch a retry a minute later would have delivered.
|
||||
RETRY_COOLDOWN_SECONDS = 60
|
||||
|
||||
|
||||
def is_enabled() -> bool:
|
||||
@@ -83,9 +123,126 @@ def is_enabled() -> bool:
|
||||
|
||||
|
||||
def _digest(value: str, length: int = 16) -> str:
|
||||
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _salt_path() -> Path:
|
||||
return memory_core.data_dir() / "telemetry-salt"
|
||||
|
||||
|
||||
def _install_salt() -> str:
|
||||
"""Random per-install salt, created once and memoized for the process.
|
||||
|
||||
Deliberately its own file, claimed with O_CREAT|O_EXCL, rather than a key in
|
||||
the identity file. Three reasons, all of which produced wrong data when this
|
||||
lived in the identity dict:
|
||||
|
||||
- Hooks are short-lived separate processes firing on every tool call, and
|
||||
people run more than one agent window. A read-modify-write would let each
|
||||
process mint its own salt, so one repository would hash several ways in the
|
||||
window before a writer won.
|
||||
- resolve_distinct_id holds a copy of the identity dict across a network call
|
||||
to /v1/ping/, so whichever write landed second erased the other's key —
|
||||
losing either the salt (repo_hash changes mid-stream) or the email (a
|
||||
second $identify, splitting the person).
|
||||
- Touching the identity file from record() would create it, and is_first_run
|
||||
keys off that file, so recording an event would silently suppress the
|
||||
install event.
|
||||
|
||||
Published atomically, and there is deliberately no derived fallback. Creating
|
||||
the file with O_CREAT|O_EXCL and then writing into it leaves a window where
|
||||
the file exists and is empty, and a concurrent hook that reads it in that
|
||||
window gets nothing. Falling back to a digest of the path would hand that
|
||||
process a salt an attacker can compute, memoized for its whole run, which is
|
||||
the privacy control this function exists to provide silently turning itself
|
||||
off under load. The salt is written to a private temp file first and linked
|
||||
into place, so the name either does not exist or already has the full value.
|
||||
|
||||
Returns "" when it genuinely cannot persist. Callers omit the hash entirely
|
||||
rather than emit an unsalted one.
|
||||
"""
|
||||
global _salt_cache
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
|
||||
path = _salt_path()
|
||||
# Read before writing. Hooks are separate processes firing on every tool
|
||||
# call, so all but the first find the salt already published; going straight
|
||||
# to create-fsync-link-unlink meant every one of them paid an fsync to
|
||||
# discover that, on a path whose whole promise is appending a line and
|
||||
# returning.
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
temporary = path.with_name(f"{path.name}.{os.getpid()}.tmp")
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
stream.write(uuid.uuid4().hex)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
try:
|
||||
# Atomic claim: fails if another process already published one.
|
||||
# os.link rather than replace, which would clobber theirs.
|
||||
os.link(temporary, path)
|
||||
except FileExistsError:
|
||||
pass
|
||||
except OSError:
|
||||
# No hardlinks here (some network mounts, some container volumes).
|
||||
# Claim the name directly instead. That reopens the empty-file
|
||||
# window, but the window is now benign: a reader that lands in it
|
||||
# gets "" and omits the hash for that process rather than caching a
|
||||
# guessable one. Losing the hashes on every run of an entire
|
||||
# filesystem is the worse failure.
|
||||
try:
|
||||
fallback = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fallback, "w", encoding="utf-8") as stream:
|
||||
stream.write(temporary.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_salt_cache = ""
|
||||
return _salt_cache
|
||||
|
||||
|
||||
def _scoped_digest(value: str, length: int = 16) -> str:
|
||||
"""Salted digest for values drawn from a guessable space.
|
||||
|
||||
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
||||
no remote — which normally contains the account username. Sixteen unsalted
|
||||
hex characters over that input space is enumerable, so this is not a
|
||||
privacy control without the salt. Salting per install keeps every
|
||||
within-account join the analytics actually use and gives up only
|
||||
cross-machine joins on the same repository, which nothing computes.
|
||||
|
||||
Returns "" when there is no salt, so record() omits the property. An
|
||||
unsalted digest over this input space is close to plaintext, and emitting one
|
||||
under a name that implies it is hashed is worse than sending nothing.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
salt = _install_salt()
|
||||
if not salt:
|
||||
return ""
|
||||
return hashlib.sha256(f"{salt}:{value}".encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _safe_value(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return memory_core.redact(value)
|
||||
@@ -145,9 +302,176 @@ def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
||||
return created
|
||||
|
||||
|
||||
def _rotate_anonymous_id(identity: dict[str, str]) -> str:
|
||||
"""Mint a fresh anonymous id because the account context is gone.
|
||||
|
||||
The previous id may already have been merged into a person profile by an
|
||||
$identify, and that merge is permanent. Reusing it after a logout or a key
|
||||
change attributes everything that follows to the account that just went
|
||||
away, which is the same misattribution the key fingerprint exists to stop,
|
||||
only arriving through the anonymous path instead.
|
||||
|
||||
`aliased` is cleared with it: the new id has never been merged, so it is
|
||||
eligible to be aliased into whatever account comes next.
|
||||
"""
|
||||
created = f"code-anon-{uuid.uuid4().hex}"
|
||||
identity["anonymous_id"] = created
|
||||
identity.pop("aliased", None)
|
||||
_write_identity(identity)
|
||||
return created
|
||||
|
||||
|
||||
def _install_state_path() -> Path:
|
||||
return memory_core.data_dir() / "install-state.json"
|
||||
|
||||
|
||||
def is_first_run() -> bool:
|
||||
"""Whether this machine has never recorded a plugin event before."""
|
||||
return not _identity_path().exists()
|
||||
"""Whether install has never been recorded on this machine.
|
||||
|
||||
Deliberately NOT the identity file. That file is only written by a
|
||||
successful flush, so an offline or firewalled user recorded code.install on
|
||||
every single session, forever — and every 0.2.x user recorded one on their
|
||||
first 0.3.x session because 0.2.x never wrote it at all.
|
||||
"""
|
||||
return not _install_state_path().exists()
|
||||
|
||||
|
||||
def data_dir_was_empty() -> bool:
|
||||
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
||||
|
||||
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
||||
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
||||
asking at claim time always saw content and every fresh install reported an
|
||||
upgrade. The caller snapshots this at the top of the run instead.
|
||||
"""
|
||||
return not _data_dir_has_content()
|
||||
|
||||
|
||||
def claim_install(was_empty: bool | None = None) -> str | None:
|
||||
"""Claim the one install/upgrade record for this machine, atomically.
|
||||
|
||||
Returns the event to record ("install" or "upgrade"), or None if another
|
||||
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
||||
cannot both win.
|
||||
|
||||
`was_empty` must come from data_dir_was_empty() called before this process
|
||||
wrote anything. Omitting it falls back to checking now, which is only
|
||||
correct for a caller that has touched nothing.
|
||||
"""
|
||||
if not is_enabled():
|
||||
# Never consume the one-shot claim while the user is opted out, or they
|
||||
# would silently lose their install event if they later opt in.
|
||||
return None
|
||||
|
||||
path = _install_state_path()
|
||||
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
json.dump(
|
||||
{
|
||||
"plugin_version": memory_core.PLUGIN_VERSION,
|
||||
"installed_at": memory_core.utc_now(),
|
||||
"upgraded": upgrading,
|
||||
},
|
||||
stream,
|
||||
)
|
||||
# Durable before this returns. The O_EXCL open is what makes the
|
||||
# claim exclusive, so it cannot be replaced by a temp-and-rename
|
||||
# without losing that, which leaves the content as the thing to make
|
||||
# safe. A kill between the open and this fsync used to leave a marker
|
||||
# that exists but parses to nothing: is_first_run reads it as claimed
|
||||
# and claim_version_change cannot read a version out of it.
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
except OSError:
|
||||
pass
|
||||
return "upgrade" if upgrading else "install"
|
||||
|
||||
|
||||
def _data_dir_has_content() -> bool:
|
||||
"""Whether anything predates this session in the plugin data directory."""
|
||||
try:
|
||||
for entry in memory_core.data_dir().iterdir():
|
||||
if entry.name != "install-state.json":
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _repair_install_state(path: Path) -> None:
|
||||
"""Rewrite an unparseable marker so version tracking can resume."""
|
||||
try:
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
temporary.write_text(
|
||||
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def claim_version_change() -> str | None:
|
||||
"""Return the previously recorded version if it differs, updating the marker.
|
||||
|
||||
Only meaningful once the marker exists — the first transition into 0.3.x has
|
||||
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
||||
the marker means the next session sees no change and records nothing.
|
||||
"""
|
||||
path = _install_state_path()
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
return None
|
||||
except json.JSONDecodeError:
|
||||
# A crash between O_EXCL and the write leaves an empty marker. Left
|
||||
# alone it disables every future upgrade event on this machine, because
|
||||
# claim_install sees the file and this function cannot parse it.
|
||||
state = None
|
||||
if not isinstance(state, dict):
|
||||
_repair_install_state(path)
|
||||
return None
|
||||
previous = str(state.get("plugin_version") or "")
|
||||
if not previous or previous == memory_core.PLUGIN_VERSION:
|
||||
return None
|
||||
# Claim the transition with an exclusive sentinel before rewriting the
|
||||
# marker. A plain read-modify-write let every concurrently starting session
|
||||
# observe the old version and each record its own upgrade — and the first
|
||||
# session after a version bump is exactly when several agent windows restart
|
||||
# together.
|
||||
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
||||
try:
|
||||
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
||||
state["upgraded_at"] = memory_core.utc_now()
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
try:
|
||||
temporary.write_text(json.dumps(state), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
# Release the claim. The marker still records the old version, so
|
||||
# without this the sentinel makes claim_version_change return early on
|
||||
# every later run and this version's upgrade is never recorded again.
|
||||
for leftover in (sentinel, temporary):
|
||||
try:
|
||||
leftover.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
return previous
|
||||
|
||||
|
||||
def record(
|
||||
@@ -168,19 +492,32 @@ def record(
|
||||
except OSError:
|
||||
pass
|
||||
properties = _safe_value(properties)
|
||||
# Stamped in the RECORDING process, beside harness. `source` used to be
|
||||
# read in the sending process from a module global, so whichever process
|
||||
# drained the spool named every event in it. flush() spreads per-event
|
||||
# properties last, so this now wins over any sender's default.
|
||||
properties.update(
|
||||
harness=_harness,
|
||||
source=_source_tag,
|
||||
plugin_version=memory_core.PLUGIN_VERSION,
|
||||
os=sys.platform,
|
||||
python_version=platform.python_version(),
|
||||
)
|
||||
# Assigned only when the digest is real. _scoped_digest returns "" when
|
||||
# the salt could not be persisted, and an empty property is worse than an
|
||||
# absent one: it survives the None filter below and reads as a value.
|
||||
if repo is not None:
|
||||
properties["repo_hash"] = _digest(getattr(repo, "identity", ""))
|
||||
repo_hash = _scoped_digest(getattr(repo, "identity", ""))
|
||||
if repo_hash:
|
||||
properties["repo_hash"] = repo_hash
|
||||
if session_id:
|
||||
properties["session_hash"] = _digest(session_id)
|
||||
session_hash = _scoped_digest(session_id)
|
||||
if session_hash:
|
||||
properties["session_hash"] = session_hash
|
||||
line = json.dumps(
|
||||
{
|
||||
"event": f"{EVENT_PREFIX}.{event}",
|
||||
"uuid": str(uuid.uuid4()),
|
||||
"timestamp": memory_core.utc_now(),
|
||||
"properties": {
|
||||
key: value for key, value in properties.items() if value is not None
|
||||
@@ -239,38 +576,201 @@ def spawn_flush() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _claim_name(attempt: int = 0) -> str:
|
||||
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
||||
only ever discards a batch that was actually retried and failed."""
|
||||
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
||||
|
||||
|
||||
def _claim_attempt(claim: Path) -> int:
|
||||
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape.
|
||||
|
||||
Anchored on field position, not on a leading "a": the legacy shape is
|
||||
``telemetry-<pid>-<hex>.sending`` and a hex id such as ``a1234567`` would
|
||||
otherwise parse as attempt 1234567 and be discarded unsent on the first
|
||||
flush after an upgrade.
|
||||
"""
|
||||
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
||||
parts = stem.split("-")
|
||||
if len(parts) != 4:
|
||||
return 0
|
||||
tail = parts[3]
|
||||
if tail.startswith("a") and tail[1:].isdigit():
|
||||
return int(tail[1:])
|
||||
return 0
|
||||
|
||||
|
||||
def _touch(path: Path) -> None:
|
||||
"""Refresh mtime so a claim's age measures time since it was claimed.
|
||||
|
||||
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
||||
after a quiet minute inherited the spool's last-write time and looked
|
||||
abandoned the instant it was made. A second sender would then take it over
|
||||
while the first was still posting, and both would deliver the batch.
|
||||
"""
|
||||
try:
|
||||
os.utime(path, None)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _claim_spool() -> Path | None:
|
||||
"""Rename the spool aside so exactly one sender owns each batch."""
|
||||
directory = memory_core.data_dir()
|
||||
claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending"
|
||||
claim = directory / _claim_name()
|
||||
spool = _spool_path()
|
||||
try:
|
||||
spool.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
pass
|
||||
return _claim_parked(directory)
|
||||
|
||||
|
||||
def _sweep_debris(directory: Path) -> None:
|
||||
"""Remove files nothing else will ever pick up again.
|
||||
|
||||
*.partial is a temp file orphaned by a crash between write and rename.
|
||||
*.corrupt is a batch quarantined for undecodable content. No glob in this
|
||||
module matches either, so without this they accumulate on disk for the life
|
||||
of the install.
|
||||
|
||||
Quarantined batches are kept far longer than debris: they are the only
|
||||
evidence left of events that could not be delivered, and someone diagnosing
|
||||
a report of missing telemetry has to be able to find one.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending")):
|
||||
for debris in directory.glob("telemetry-*.partial"):
|
||||
try:
|
||||
if now - debris.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
debris.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
for quarantined in directory.glob("telemetry-*.corrupt"):
|
||||
try:
|
||||
if now - quarantined.stat().st_mtime > CLAIM_EXPIRY_SECONDS:
|
||||
quarantined.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
# The same reasoning covers *.tmp. _write_identity and _install_salt both
|
||||
# create one and unlink it in a finally, which a SIGKILL skips, and no glob
|
||||
# in this module matches the leftovers either.
|
||||
for temporary in directory.glob("telemetry-*.tmp"):
|
||||
try:
|
||||
if now - temporary.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
|
||||
def _claim_parked(directory: Path) -> Path | None:
|
||||
"""Take the oldest abandoned claim, if any lease has actually expired.
|
||||
|
||||
Kept separate from the live spool so flush() can drain both in one run.
|
||||
Previously parked batches were only reachable when no spool existed at all,
|
||||
and because sessions keep recording there usually was one — so a batch
|
||||
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
||||
even though its own presence is what started the sender.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
||||
try:
|
||||
age = now - orphan.stat().st_mtime
|
||||
except OSError:
|
||||
continue
|
||||
if age > CLAIM_EXPIRY_SECONDS:
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
# Someone else holds a live lease on it. This check has to come
|
||||
# first. Claiming a file bumps its attempt count and refreshes its
|
||||
# mtime, so a sender that has just taken the final attempt looks
|
||||
# exhausted to everyone else while it is actively draining. Judging
|
||||
# exhaustion before liveness let a second sender unlink a batch out
|
||||
# from under its owner, losing every event in it.
|
||||
continue
|
||||
# Attempts, not age. Every re-claim touches the mtime and every release
|
||||
# backdates it by a fixed amount, so age is pinned near the stale
|
||||
# threshold and never reaches the expiry. Age stays only as a backstop
|
||||
# for files that never carried an attempt marker.
|
||||
if _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS or age > CLAIM_EXPIRY_SECONDS:
|
||||
try:
|
||||
orphan.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
continue
|
||||
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
||||
try:
|
||||
orphan.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _safe_mtime(path: Path) -> float:
|
||||
try:
|
||||
return path.stat().st_mtime
|
||||
except OSError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
||||
"""Persist the unsent remainder, atomically, and refresh the lease.
|
||||
|
||||
Called after every successful batch. Two jobs: a retry resumes where the
|
||||
send stopped instead of re-posting from the top, and the rewrite doubles as
|
||||
the lease heartbeat, so a slow sender does not have its claim stolen
|
||||
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
||||
"""
|
||||
if not remaining:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return True
|
||||
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
||||
try:
|
||||
payload = "".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining)
|
||||
# fsync before the rename: without it the rename can land while the
|
||||
# bytes have not, and the claim comes back empty or truncated after a
|
||||
# crash. _drain then reads zero events and unlinks it.
|
||||
with open(temporary, "w", encoding="utf-8") as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
temporary.replace(claim)
|
||||
_touch(claim)
|
||||
return True
|
||||
except OSError:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
||||
"""Persist the remainder and drop the lease, because this sender has given up.
|
||||
|
||||
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
||||
make an abandoned batch look actively owned for a further
|
||||
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
||||
threshold lets the next flush pick it up immediately, while the attempt
|
||||
count in the filename still bounds how many times that can happen.
|
||||
"""
|
||||
if not _rewrite_claim(claim, remaining):
|
||||
return
|
||||
try:
|
||||
# Backdate past the stale threshold so the next flush can pick it up,
|
||||
# minus a cooldown that grows with the attempts already spent. Clamped so
|
||||
# the mtime never lands in the future, which would read as a live lease.
|
||||
cooldown = min(_claim_attempt(claim) * RETRY_COOLDOWN_SECONDS, CLAIM_STALE_SECONDS)
|
||||
released = time.time() - CLAIM_STALE_SECONDS - 1 + cooldown
|
||||
os.utime(claim, (released, released))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_email(key: str) -> str:
|
||||
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
||||
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
||||
@@ -300,34 +800,130 @@ def _post(payload: dict[str, Any], url: str) -> bool:
|
||||
|
||||
|
||||
def resolve_distinct_id() -> tuple[str, str]:
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any."""
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
||||
|
||||
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
||||
anonymous id: aliasing one account email to another merges two real person
|
||||
profiles and cannot be undone, so a key that now belongs to a different
|
||||
account re-resolves with no alias.
|
||||
"""
|
||||
identity = _read_identity()
|
||||
email = identity.get("email", "")
|
||||
if email:
|
||||
return email, ""
|
||||
key = memory_core.api_key()
|
||||
fingerprint = _digest(key) if key else ""
|
||||
email = identity.get("email", "")
|
||||
|
||||
if email and fingerprint:
|
||||
recorded = identity.get("key_fingerprint", "")
|
||||
if recorded == fingerprint:
|
||||
return email, ""
|
||||
if not recorded:
|
||||
# Rows written before fingerprints existed. Verify rather than
|
||||
# adopt: a key changed before the upgrade would otherwise bind the
|
||||
# new key to the previous account's email, permanently, and the
|
||||
# fingerprint would then agree with itself forever after.
|
||||
verified = _resolve_email(key)
|
||||
if not verified:
|
||||
# Offline, firewalled, or the API is down. Keep the previous
|
||||
# behaviour and retry on the next flush rather than dropping a
|
||||
# real account attribution. Safe because the same network that
|
||||
# failed /v1/ping/ is about to fail the PostHog POST, so nothing
|
||||
# is delivered under the unverified identity in the meantime.
|
||||
return email, ""
|
||||
identity["email"] = verified
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return verified, ""
|
||||
|
||||
if not key:
|
||||
# No key to verify the account with; do not keep attributing to it.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
email = _resolve_email(key)
|
||||
if not email:
|
||||
|
||||
resolved = _resolve_email(key)
|
||||
if not resolved:
|
||||
# The key changed and will not resolve (revoked, offline, API down).
|
||||
# Reaching here with an email means the recorded fingerprint disagreed,
|
||||
# so the key really did change. Drop the account and rotate: the stored
|
||||
# anonymous id may already be merged into that account's person, and
|
||||
# reusing it would keep the events on the profile we are trying to
|
||||
# leave.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
previous = identity.get("anonymous_id", "")
|
||||
identity["email"] = email
|
||||
|
||||
# Alias only when going anonymous -> email for the first time. Once an anon
|
||||
# id has been merged into an account it must never be offered again: an
|
||||
# alias naming an already-identified id is what could link two real people.
|
||||
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
||||
if previous:
|
||||
identity["aliased"] = True
|
||||
identity["email"] = resolved
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return email, previous
|
||||
return resolved, previous
|
||||
|
||||
|
||||
def flush() -> int:
|
||||
"""Drain claimed spools to PostHog and return the number of events sent."""
|
||||
"""Drain the live spool, then any parked claims, and return events sent."""
|
||||
if not is_enabled():
|
||||
return 0
|
||||
claim = _claim_spool()
|
||||
sent, delivered = _drain(_claim_spool())
|
||||
if not delivered:
|
||||
# The network is failing. Retrying other batches now would only burn
|
||||
# their attempt budget against the same broken connection.
|
||||
return sent
|
||||
|
||||
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
||||
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
||||
directory = memory_core.data_dir()
|
||||
_sweep_debris(directory)
|
||||
for _ in range(MAX_PARKED_PER_RUN):
|
||||
parked = _claim_parked(directory)
|
||||
if parked is None:
|
||||
break
|
||||
count, delivered = _drain(parked)
|
||||
sent += count
|
||||
if not delivered:
|
||||
break
|
||||
return sent
|
||||
|
||||
|
||||
def _drain(claim: Path | None) -> tuple[int, bool]:
|
||||
"""Post one claimed batch file, recording progress after every batch.
|
||||
|
||||
Returns (events sent, whether everything was delivered).
|
||||
"""
|
||||
if claim is None:
|
||||
return 0
|
||||
return 0, True
|
||||
try:
|
||||
lines = claim.read_text(encoding="utf-8").splitlines()
|
||||
except ValueError:
|
||||
# UnicodeDecodeError from a torn write: the content is unrecoverable, so
|
||||
# quarantine rather than retry. flush() runs from a bare `finally:` in
|
||||
# flush_worker, so raising here also skips the handoff cleanup, and an
|
||||
# undecodable file would otherwise be re-read on every flush forever.
|
||||
# Reported as delivered because there is nothing left to deliver and the
|
||||
# rest of the run should continue.
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt"))
|
||||
except OSError:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0, True
|
||||
except OSError:
|
||||
return 0
|
||||
# Could not read it, which is not the same as having nothing to send.
|
||||
# The file is left exactly where it is: a vanished or briefly unreadable
|
||||
# claim is retryable, and quarantining it here would discard events over
|
||||
# a transient filesystem error. Reported as undelivered so the run stops
|
||||
# instead of counting a batch nothing was posted from as delivered.
|
||||
return 0, False
|
||||
events = []
|
||||
for line in lines:
|
||||
try:
|
||||
@@ -337,11 +933,18 @@ def flush() -> int:
|
||||
if isinstance(value, dict) and value.get("event"):
|
||||
events.append(value)
|
||||
if not events:
|
||||
# Only delete when the file really is empty. A non-empty file that
|
||||
# parses to nothing is a torn write, and its contents are the unsent
|
||||
# remainder — deleting it is the data loss this PR exists to prevent.
|
||||
try:
|
||||
claim.unlink()
|
||||
empty = claim.stat().st_size == 0
|
||||
except OSError:
|
||||
empty = True
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt")) if not empty else claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
return 0, True
|
||||
|
||||
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
||||
if aliased_anonymous_id:
|
||||
@@ -360,12 +963,17 @@ def flush() -> int:
|
||||
|
||||
sent = 0
|
||||
for start in range(0, len(events), BATCH_SIZE):
|
||||
chunk = events[start : start + BATCH_SIZE]
|
||||
batch = [
|
||||
{
|
||||
"event": event["event"],
|
||||
"distinct_id": distinct_id,
|
||||
# Carried through from record() so a resend can be collapsed.
|
||||
"uuid": event.get("uuid"),
|
||||
"timestamp": event.get("timestamp"),
|
||||
"properties": {
|
||||
# Fallback only: events recorded by a build before source
|
||||
# moved into record() have none of their own.
|
||||
"source": _source_tag,
|
||||
"language": "python",
|
||||
"$process_person_profile": False,
|
||||
@@ -373,16 +981,24 @@ def flush() -> int:
|
||||
**(event.get("properties") or {}),
|
||||
},
|
||||
}
|
||||
for event in events[start : start + BATCH_SIZE]
|
||||
for event in chunk
|
||||
]
|
||||
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
||||
return sent
|
||||
sent += len(batch)
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return sent
|
||||
# Keep only what has not been delivered, and release the lease.
|
||||
# Previously the whole file was kept and the retry re-posted every
|
||||
# batch, including the ones that had already arrived.
|
||||
_release_claim(claim, events[start:])
|
||||
return sent, False
|
||||
sent += len(chunk)
|
||||
# Record progress and refresh the lease after each successful batch, so
|
||||
# a crash repeats at most one batch instead of the entire file. If the
|
||||
# rewrite fails the claim still holds delivered events, so stop rather
|
||||
# than carry on as though progress were recorded — continuing is how the
|
||||
# duplicate delivery this PR fixes would come back.
|
||||
if not _rewrite_claim(claim, events[start + len(chunk) :]):
|
||||
_release_claim(claim, events[start + len(chunk) :])
|
||||
return sent, False
|
||||
return sent, True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"keywords": ["memory", "coding-agents", "continual-learning", "token-efficiency"],
|
||||
"author": { "name": "Mem0", "email": "support@mem0.ai" },
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"id": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"homepage": "https://docs.mem0.ai/integrations/kimi",
|
||||
"native": {
|
||||
"pluginRoot": "${KIMI_PLUGIN_ROOT}",
|
||||
|
||||
@@ -7,8 +7,8 @@ disable-model-invocation: true
|
||||
# Pause memory capture
|
||||
|
||||
To pause (hooks stop capturing and sending session content; a minimal
|
||||
anonymous telemetry ping still fires at session start unless
|
||||
`MEM0_TELEMETRY=false`):
|
||||
telemetry ping still fires at session start, under your Mem0 account email,
|
||||
unless `MEM0_TELEMETRY=false`):
|
||||
|
||||
```bash
|
||||
python3 "${KIMI_PLUGIN_ROOT}/core/memory_cli.py" --harness "kimi" pause
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Generated by integrations/agent-plugin-core/build/build.py. Do not edit."""
|
||||
|
||||
HARNESS_ID = "coding-agent"
|
||||
SOURCE_TAG = "CODING_AGENT_PLUGIN"
|
||||
|
||||
# Platform-side vocabulary (mem0_event.source + X-Application). The whole
|
||||
# plugin family is one source; which editor it runs in is the application.
|
||||
# An empty application means the host is unknown, and memory_core omits
|
||||
# the header entirely rather than sending a placeholder.
|
||||
PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
PLATFORM_APPLICATION = ""
|
||||
@@ -29,7 +29,7 @@ from typing import Any, Iterable
|
||||
import telemetry
|
||||
|
||||
DEFAULT_API_URL = "https://api.mem0.ai"
|
||||
PLUGIN_VERSION = "0.3.1"
|
||||
PLUGIN_VERSION = "0.3.2"
|
||||
|
||||
_harness_name: str = "generic"
|
||||
_harness_env_prefix: str = "MEM0_PLUGIN"
|
||||
@@ -1800,6 +1800,34 @@ def extraction_message_batches(
|
||||
return batches
|
||||
|
||||
|
||||
# Platform surface attribution. Read from the generated per-host module so a new
|
||||
# entrypoint is correct without remembering to configure anything.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
except ImportError:
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
|
||||
def platform_headers(key: str) -> dict[str, str]:
|
||||
"""Auth plus the three surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are set-once by contract: this is the
|
||||
outermost layer, so it sets them, and nothing below may overwrite them.
|
||||
X-Mem0-Client is append-only — anything downstream adds itself to the tail.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {key}",
|
||||
"Content-Type": "application/json",
|
||||
"X-Mem0-Source": _PLATFORM_SOURCE,
|
||||
"X-Mem0-Client": f"mem0-plugin/{PLUGIN_VERSION}",
|
||||
}
|
||||
if _PLATFORM_APPLICATION:
|
||||
headers["X-Application"] = _PLATFORM_APPLICATION
|
||||
return headers
|
||||
|
||||
|
||||
def _request_json(
|
||||
url: str, key: str, payload: dict[str, Any], timeout: float
|
||||
) -> tuple[dict[str, Any] | list[Any], int, int]:
|
||||
@@ -1807,7 +1835,7 @@ def _request_json(
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1834,7 +1862,7 @@ def _get_json(
|
||||
) -> tuple[dict[str, Any] | list[Any], int]:
|
||||
request = urllib.request.Request(
|
||||
url,
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="GET",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
@@ -1980,6 +2008,13 @@ def flush_session(
|
||||
"user_id": write_user,
|
||||
"app_id": repo.app_id,
|
||||
"run_id": session_id,
|
||||
# Top level, not metadata: the backend reads `source` from the body or
|
||||
# the query string, never from metadata, which is where this used to
|
||||
# sit. The X-Mem0-Source header is also read, but only from the
|
||||
# platform release that ships alongside this change, so the body value
|
||||
# is what makes attribution work on both. The harness tag stays in
|
||||
# metadata as hook provenance.
|
||||
"source": _PLATFORM_SOURCE,
|
||||
"metadata": {**metadata, "author": write_user, "dirs": directory_chain(repo)},
|
||||
"agent_custom_instructions": PROJECT_MEMORY_INSTRUCTIONS,
|
||||
"custom_instructions": PERSONAL_MEMORY_INSTRUCTIONS,
|
||||
@@ -2523,7 +2558,7 @@ def _collect_memory_ids(
|
||||
def _delete_memory(api_url: str, key: str, memory_id: str) -> bool:
|
||||
request = urllib.request.Request(
|
||||
f"{api_url}/v1/memories/{urllib.parse.quote(memory_id)}/",
|
||||
headers={"Authorization": f"Token {key}", "Content-Type": "application/json"},
|
||||
headers=platform_headers(key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Anonymous usage telemetry for Mem0 agent plugins.
|
||||
"""Usage telemetry for Mem0 agent plugins.
|
||||
|
||||
Events are linked to your Mem0 account email when an API key is configured, and
|
||||
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
||||
attribute the same way.
|
||||
|
||||
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
||||
touches the network: `record` appends one JSON line to a local spool and returns.
|
||||
@@ -9,7 +13,8 @@ started once per session and again from the flush worker that is already detache
|
||||
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
||||
|
||||
Never sends prompts, memory text, queries, file paths, repository names, or API
|
||||
keys: only event names, durations, counts, coarse outcomes, and salted hashes.
|
||||
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
||||
identifiers hashed with a random per-install salt.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -29,8 +34,24 @@ from typing import Any
|
||||
|
||||
import memory_core
|
||||
|
||||
_harness: str = "generic"
|
||||
_source_tag: str = "MEM0_PLUGIN"
|
||||
# Seeded from the per-host module the build generates into core/. Two processes
|
||||
# in this pipeline never call init() — mcp_server.py, and the detached
|
||||
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
||||
# was what every one of their events got labelled with.
|
||||
try: # pragma: no cover - absent only in the un-built shared source tree
|
||||
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
||||
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
||||
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
||||
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
||||
except ImportError:
|
||||
_DEFAULT_HARNESS = "generic"
|
||||
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
||||
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
||||
_PLATFORM_APPLICATION = ""
|
||||
|
||||
_salt_cache: str = ""
|
||||
_harness: str = _DEFAULT_HARNESS
|
||||
_source_tag: str = _DEFAULT_SOURCE_TAG
|
||||
_PRIVATE_KEYS = {
|
||||
"apikey",
|
||||
"authorization",
|
||||
@@ -56,10 +77,19 @@ _PRIVATE_KEYS = {
|
||||
}
|
||||
|
||||
|
||||
def init(harness: str = "generic", source_tag: str = "") -> None:
|
||||
def init(harness: str = "", source_tag: str = "") -> None:
|
||||
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
||||
|
||||
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
||||
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
||||
meant one plugin could emit three different source values depending on which
|
||||
process happened to send the batch.
|
||||
"""
|
||||
global _harness, _source_tag
|
||||
_harness = harness
|
||||
_source_tag = source_tag or f"MEM0_{harness.upper().replace('-', '_')}_PLUGIN"
|
||||
_harness = harness or _DEFAULT_HARNESS
|
||||
_source_tag = source_tag or (
|
||||
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
||||
)
|
||||
|
||||
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
||||
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
||||
@@ -70,6 +100,16 @@ BATCH_SIZE = 100
|
||||
SEND_TIMEOUT = 5
|
||||
CLAIM_STALE_SECONDS = 120
|
||||
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
||||
# A batch is only discarded once it has genuinely been retried this many times.
|
||||
MAX_CLAIM_ATTEMPTS = 3
|
||||
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
||||
# cannot turn one flush into an unbounded send loop.
|
||||
MAX_PARKED_PER_RUN = 3
|
||||
# Added to the wait before a released claim becomes reclaimable, per attempt
|
||||
# already spent. Releasing straight to "reclaimable now" let two senders burn the
|
||||
# whole budget within seconds of one another on a single momentary failure, and
|
||||
# discard a batch a retry a minute later would have delivered.
|
||||
RETRY_COOLDOWN_SECONDS = 60
|
||||
|
||||
|
||||
def is_enabled() -> bool:
|
||||
@@ -83,9 +123,126 @@ def is_enabled() -> bool:
|
||||
|
||||
|
||||
def _digest(value: str, length: int = 16) -> str:
|
||||
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _salt_path() -> Path:
|
||||
return memory_core.data_dir() / "telemetry-salt"
|
||||
|
||||
|
||||
def _install_salt() -> str:
|
||||
"""Random per-install salt, created once and memoized for the process.
|
||||
|
||||
Deliberately its own file, claimed with O_CREAT|O_EXCL, rather than a key in
|
||||
the identity file. Three reasons, all of which produced wrong data when this
|
||||
lived in the identity dict:
|
||||
|
||||
- Hooks are short-lived separate processes firing on every tool call, and
|
||||
people run more than one agent window. A read-modify-write would let each
|
||||
process mint its own salt, so one repository would hash several ways in the
|
||||
window before a writer won.
|
||||
- resolve_distinct_id holds a copy of the identity dict across a network call
|
||||
to /v1/ping/, so whichever write landed second erased the other's key —
|
||||
losing either the salt (repo_hash changes mid-stream) or the email (a
|
||||
second $identify, splitting the person).
|
||||
- Touching the identity file from record() would create it, and is_first_run
|
||||
keys off that file, so recording an event would silently suppress the
|
||||
install event.
|
||||
|
||||
Published atomically, and there is deliberately no derived fallback. Creating
|
||||
the file with O_CREAT|O_EXCL and then writing into it leaves a window where
|
||||
the file exists and is empty, and a concurrent hook that reads it in that
|
||||
window gets nothing. Falling back to a digest of the path would hand that
|
||||
process a salt an attacker can compute, memoized for its whole run, which is
|
||||
the privacy control this function exists to provide silently turning itself
|
||||
off under load. The salt is written to a private temp file first and linked
|
||||
into place, so the name either does not exist or already has the full value.
|
||||
|
||||
Returns "" when it genuinely cannot persist. Callers omit the hash entirely
|
||||
rather than emit an unsalted one.
|
||||
"""
|
||||
global _salt_cache
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
|
||||
path = _salt_path()
|
||||
# Read before writing. Hooks are separate processes firing on every tool
|
||||
# call, so all but the first find the salt already published; going straight
|
||||
# to create-fsync-link-unlink meant every one of them paid an fsync to
|
||||
# discover that, on a path whose whole promise is appending a line and
|
||||
# returning.
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
if _salt_cache:
|
||||
return _salt_cache
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
temporary = path.with_name(f"{path.name}.{os.getpid()}.tmp")
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(temporary, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
stream.write(uuid.uuid4().hex)
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
try:
|
||||
# Atomic claim: fails if another process already published one.
|
||||
# os.link rather than replace, which would clobber theirs.
|
||||
os.link(temporary, path)
|
||||
except FileExistsError:
|
||||
pass
|
||||
except OSError:
|
||||
# No hardlinks here (some network mounts, some container volumes).
|
||||
# Claim the name directly instead. That reopens the empty-file
|
||||
# window, but the window is now benign: a reader that lands in it
|
||||
# gets "" and omits the hash for that process rather than caching a
|
||||
# guessable one. Losing the hashes on every run of an entire
|
||||
# filesystem is the worse failure.
|
||||
try:
|
||||
fallback = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
with os.fdopen(fallback, "w", encoding="utf-8") as stream:
|
||||
stream.write(temporary.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
_salt_cache = path.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
_salt_cache = ""
|
||||
return _salt_cache
|
||||
|
||||
|
||||
def _scoped_digest(value: str, length: int = 16) -> str:
|
||||
"""Salted digest for values drawn from a guessable space.
|
||||
|
||||
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
||||
no remote — which normally contains the account username. Sixteen unsalted
|
||||
hex characters over that input space is enumerable, so this is not a
|
||||
privacy control without the salt. Salting per install keeps every
|
||||
within-account join the analytics actually use and gives up only
|
||||
cross-machine joins on the same repository, which nothing computes.
|
||||
|
||||
Returns "" when there is no salt, so record() omits the property. An
|
||||
unsalted digest over this input space is close to plaintext, and emitting one
|
||||
under a name that implies it is hashed is worse than sending nothing.
|
||||
"""
|
||||
if not value:
|
||||
return ""
|
||||
salt = _install_salt()
|
||||
if not salt:
|
||||
return ""
|
||||
return hashlib.sha256(f"{salt}:{value}".encode("utf-8")).hexdigest()[:length]
|
||||
|
||||
|
||||
def _safe_value(value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return memory_core.redact(value)
|
||||
@@ -145,9 +302,176 @@ def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
||||
return created
|
||||
|
||||
|
||||
def _rotate_anonymous_id(identity: dict[str, str]) -> str:
|
||||
"""Mint a fresh anonymous id because the account context is gone.
|
||||
|
||||
The previous id may already have been merged into a person profile by an
|
||||
$identify, and that merge is permanent. Reusing it after a logout or a key
|
||||
change attributes everything that follows to the account that just went
|
||||
away, which is the same misattribution the key fingerprint exists to stop,
|
||||
only arriving through the anonymous path instead.
|
||||
|
||||
`aliased` is cleared with it: the new id has never been merged, so it is
|
||||
eligible to be aliased into whatever account comes next.
|
||||
"""
|
||||
created = f"code-anon-{uuid.uuid4().hex}"
|
||||
identity["anonymous_id"] = created
|
||||
identity.pop("aliased", None)
|
||||
_write_identity(identity)
|
||||
return created
|
||||
|
||||
|
||||
def _install_state_path() -> Path:
|
||||
return memory_core.data_dir() / "install-state.json"
|
||||
|
||||
|
||||
def is_first_run() -> bool:
|
||||
"""Whether this machine has never recorded a plugin event before."""
|
||||
return not _identity_path().exists()
|
||||
"""Whether install has never been recorded on this machine.
|
||||
|
||||
Deliberately NOT the identity file. That file is only written by a
|
||||
successful flush, so an offline or firewalled user recorded code.install on
|
||||
every single session, forever — and every 0.2.x user recorded one on their
|
||||
first 0.3.x session because 0.2.x never wrote it at all.
|
||||
"""
|
||||
return not _install_state_path().exists()
|
||||
|
||||
|
||||
def data_dir_was_empty() -> bool:
|
||||
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
||||
|
||||
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
||||
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
||||
asking at claim time always saw content and every fresh install reported an
|
||||
upgrade. The caller snapshots this at the top of the run instead.
|
||||
"""
|
||||
return not _data_dir_has_content()
|
||||
|
||||
|
||||
def claim_install(was_empty: bool | None = None) -> str | None:
|
||||
"""Claim the one install/upgrade record for this machine, atomically.
|
||||
|
||||
Returns the event to record ("install" or "upgrade"), or None if another
|
||||
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
||||
cannot both win.
|
||||
|
||||
`was_empty` must come from data_dir_was_empty() called before this process
|
||||
wrote anything. Omitting it falls back to checking now, which is only
|
||||
correct for a caller that has touched nothing.
|
||||
"""
|
||||
if not is_enabled():
|
||||
# Never consume the one-shot claim while the user is opted out, or they
|
||||
# would silently lose their install event if they later opt in.
|
||||
return None
|
||||
|
||||
path = _install_state_path()
|
||||
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
try:
|
||||
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
||||
json.dump(
|
||||
{
|
||||
"plugin_version": memory_core.PLUGIN_VERSION,
|
||||
"installed_at": memory_core.utc_now(),
|
||||
"upgraded": upgrading,
|
||||
},
|
||||
stream,
|
||||
)
|
||||
# Durable before this returns. The O_EXCL open is what makes the
|
||||
# claim exclusive, so it cannot be replaced by a temp-and-rename
|
||||
# without losing that, which leaves the content as the thing to make
|
||||
# safe. A kill between the open and this fsync used to leave a marker
|
||||
# that exists but parses to nothing: is_first_run reads it as claimed
|
||||
# and claim_version_change cannot read a version out of it.
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
except OSError:
|
||||
pass
|
||||
return "upgrade" if upgrading else "install"
|
||||
|
||||
|
||||
def _data_dir_has_content() -> bool:
|
||||
"""Whether anything predates this session in the plugin data directory."""
|
||||
try:
|
||||
for entry in memory_core.data_dir().iterdir():
|
||||
if entry.name != "install-state.json":
|
||||
return True
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _repair_install_state(path: Path) -> None:
|
||||
"""Rewrite an unparseable marker so version tracking can resume."""
|
||||
try:
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
temporary.write_text(
|
||||
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def claim_version_change() -> str | None:
|
||||
"""Return the previously recorded version if it differs, updating the marker.
|
||||
|
||||
Only meaningful once the marker exists — the first transition into 0.3.x has
|
||||
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
||||
the marker means the next session sees no change and records nothing.
|
||||
"""
|
||||
path = _install_state_path()
|
||||
try:
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
except OSError:
|
||||
return None
|
||||
except json.JSONDecodeError:
|
||||
# A crash between O_EXCL and the write leaves an empty marker. Left
|
||||
# alone it disables every future upgrade event on this machine, because
|
||||
# claim_install sees the file and this function cannot parse it.
|
||||
state = None
|
||||
if not isinstance(state, dict):
|
||||
_repair_install_state(path)
|
||||
return None
|
||||
previous = str(state.get("plugin_version") or "")
|
||||
if not previous or previous == memory_core.PLUGIN_VERSION:
|
||||
return None
|
||||
# Claim the transition with an exclusive sentinel before rewriting the
|
||||
# marker. A plain read-modify-write let every concurrently starting session
|
||||
# observe the old version and each record its own upgrade — and the first
|
||||
# session after a version bump is exactly when several agent windows restart
|
||||
# together.
|
||||
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
||||
try:
|
||||
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
||||
except FileExistsError:
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
||||
state["upgraded_at"] = memory_core.utc_now()
|
||||
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
||||
try:
|
||||
temporary.write_text(json.dumps(state), encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
except OSError:
|
||||
# Release the claim. The marker still records the old version, so
|
||||
# without this the sentinel makes claim_version_change return early on
|
||||
# every later run and this version's upgrade is never recorded again.
|
||||
for leftover in (sentinel, temporary):
|
||||
try:
|
||||
leftover.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
return previous
|
||||
|
||||
|
||||
def record(
|
||||
@@ -168,19 +492,32 @@ def record(
|
||||
except OSError:
|
||||
pass
|
||||
properties = _safe_value(properties)
|
||||
# Stamped in the RECORDING process, beside harness. `source` used to be
|
||||
# read in the sending process from a module global, so whichever process
|
||||
# drained the spool named every event in it. flush() spreads per-event
|
||||
# properties last, so this now wins over any sender's default.
|
||||
properties.update(
|
||||
harness=_harness,
|
||||
source=_source_tag,
|
||||
plugin_version=memory_core.PLUGIN_VERSION,
|
||||
os=sys.platform,
|
||||
python_version=platform.python_version(),
|
||||
)
|
||||
# Assigned only when the digest is real. _scoped_digest returns "" when
|
||||
# the salt could not be persisted, and an empty property is worse than an
|
||||
# absent one: it survives the None filter below and reads as a value.
|
||||
if repo is not None:
|
||||
properties["repo_hash"] = _digest(getattr(repo, "identity", ""))
|
||||
repo_hash = _scoped_digest(getattr(repo, "identity", ""))
|
||||
if repo_hash:
|
||||
properties["repo_hash"] = repo_hash
|
||||
if session_id:
|
||||
properties["session_hash"] = _digest(session_id)
|
||||
session_hash = _scoped_digest(session_id)
|
||||
if session_hash:
|
||||
properties["session_hash"] = session_hash
|
||||
line = json.dumps(
|
||||
{
|
||||
"event": f"{EVENT_PREFIX}.{event}",
|
||||
"uuid": str(uuid.uuid4()),
|
||||
"timestamp": memory_core.utc_now(),
|
||||
"properties": {
|
||||
key: value for key, value in properties.items() if value is not None
|
||||
@@ -239,38 +576,201 @@ def spawn_flush() -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _claim_name(attempt: int = 0) -> str:
|
||||
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
||||
only ever discards a batch that was actually retried and failed."""
|
||||
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
||||
|
||||
|
||||
def _claim_attempt(claim: Path) -> int:
|
||||
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape.
|
||||
|
||||
Anchored on field position, not on a leading "a": the legacy shape is
|
||||
``telemetry-<pid>-<hex>.sending`` and a hex id such as ``a1234567`` would
|
||||
otherwise parse as attempt 1234567 and be discarded unsent on the first
|
||||
flush after an upgrade.
|
||||
"""
|
||||
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
||||
parts = stem.split("-")
|
||||
if len(parts) != 4:
|
||||
return 0
|
||||
tail = parts[3]
|
||||
if tail.startswith("a") and tail[1:].isdigit():
|
||||
return int(tail[1:])
|
||||
return 0
|
||||
|
||||
|
||||
def _touch(path: Path) -> None:
|
||||
"""Refresh mtime so a claim's age measures time since it was claimed.
|
||||
|
||||
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
||||
after a quiet minute inherited the spool's last-write time and looked
|
||||
abandoned the instant it was made. A second sender would then take it over
|
||||
while the first was still posting, and both would deliver the batch.
|
||||
"""
|
||||
try:
|
||||
os.utime(path, None)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _claim_spool() -> Path | None:
|
||||
"""Rename the spool aside so exactly one sender owns each batch."""
|
||||
directory = memory_core.data_dir()
|
||||
claim = directory / f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}.sending"
|
||||
claim = directory / _claim_name()
|
||||
spool = _spool_path()
|
||||
try:
|
||||
spool.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
pass
|
||||
return _claim_parked(directory)
|
||||
|
||||
|
||||
def _sweep_debris(directory: Path) -> None:
|
||||
"""Remove files nothing else will ever pick up again.
|
||||
|
||||
*.partial is a temp file orphaned by a crash between write and rename.
|
||||
*.corrupt is a batch quarantined for undecodable content. No glob in this
|
||||
module matches either, so without this they accumulate on disk for the life
|
||||
of the install.
|
||||
|
||||
Quarantined batches are kept far longer than debris: they are the only
|
||||
evidence left of events that could not be delivered, and someone diagnosing
|
||||
a report of missing telemetry has to be able to find one.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending")):
|
||||
for debris in directory.glob("telemetry-*.partial"):
|
||||
try:
|
||||
if now - debris.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
debris.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
for quarantined in directory.glob("telemetry-*.corrupt"):
|
||||
try:
|
||||
if now - quarantined.stat().st_mtime > CLAIM_EXPIRY_SECONDS:
|
||||
quarantined.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
# The same reasoning covers *.tmp. _write_identity and _install_salt both
|
||||
# create one and unlink it in a finally, which a SIGKILL skips, and no glob
|
||||
# in this module matches the leftovers either.
|
||||
for temporary in directory.glob("telemetry-*.tmp"):
|
||||
try:
|
||||
if now - temporary.stat().st_mtime > CLAIM_STALE_SECONDS:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
|
||||
def _claim_parked(directory: Path) -> Path | None:
|
||||
"""Take the oldest abandoned claim, if any lease has actually expired.
|
||||
|
||||
Kept separate from the live spool so flush() can drain both in one run.
|
||||
Previously parked batches were only reachable when no spool existed at all,
|
||||
and because sessions keep recording there usually was one — so a batch
|
||||
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
||||
even though its own presence is what started the sender.
|
||||
"""
|
||||
now = time.time()
|
||||
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
||||
try:
|
||||
age = now - orphan.stat().st_mtime
|
||||
except OSError:
|
||||
continue
|
||||
if age > CLAIM_EXPIRY_SECONDS:
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
# Someone else holds a live lease on it. This check has to come
|
||||
# first. Claiming a file bumps its attempt count and refreshes its
|
||||
# mtime, so a sender that has just taken the final attempt looks
|
||||
# exhausted to everyone else while it is actively draining. Judging
|
||||
# exhaustion before liveness let a second sender unlink a batch out
|
||||
# from under its owner, losing every event in it.
|
||||
continue
|
||||
# Attempts, not age. Every re-claim touches the mtime and every release
|
||||
# backdates it by a fixed amount, so age is pinned near the stale
|
||||
# threshold and never reaches the expiry. Age stays only as a backstop
|
||||
# for files that never carried an attempt marker.
|
||||
if _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS or age > CLAIM_EXPIRY_SECONDS:
|
||||
try:
|
||||
orphan.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if age < CLAIM_STALE_SECONDS:
|
||||
continue
|
||||
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
||||
try:
|
||||
orphan.replace(claim)
|
||||
_touch(claim)
|
||||
return claim
|
||||
except OSError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def _safe_mtime(path: Path) -> float:
|
||||
try:
|
||||
return path.stat().st_mtime
|
||||
except OSError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
||||
"""Persist the unsent remainder, atomically, and refresh the lease.
|
||||
|
||||
Called after every successful batch. Two jobs: a retry resumes where the
|
||||
send stopped instead of re-posting from the top, and the rewrite doubles as
|
||||
the lease heartbeat, so a slow sender does not have its claim stolen
|
||||
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
||||
"""
|
||||
if not remaining:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return True
|
||||
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
||||
try:
|
||||
payload = "".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining)
|
||||
# fsync before the rename: without it the rename can land while the
|
||||
# bytes have not, and the claim comes back empty or truncated after a
|
||||
# crash. _drain then reads zero events and unlinks it.
|
||||
with open(temporary, "w", encoding="utf-8") as handle:
|
||||
handle.write(payload)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
temporary.replace(claim)
|
||||
_touch(claim)
|
||||
return True
|
||||
except OSError:
|
||||
try:
|
||||
temporary.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
||||
"""Persist the remainder and drop the lease, because this sender has given up.
|
||||
|
||||
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
||||
make an abandoned batch look actively owned for a further
|
||||
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
||||
threshold lets the next flush pick it up immediately, while the attempt
|
||||
count in the filename still bounds how many times that can happen.
|
||||
"""
|
||||
if not _rewrite_claim(claim, remaining):
|
||||
return
|
||||
try:
|
||||
# Backdate past the stale threshold so the next flush can pick it up,
|
||||
# minus a cooldown that grows with the attempts already spent. Clamped so
|
||||
# the mtime never lands in the future, which would read as a live lease.
|
||||
cooldown = min(_claim_attempt(claim) * RETRY_COOLDOWN_SECONDS, CLAIM_STALE_SECONDS)
|
||||
released = time.time() - CLAIM_STALE_SECONDS - 1 + cooldown
|
||||
os.utime(claim, (released, released))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def _resolve_email(key: str) -> str:
|
||||
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
||||
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
||||
@@ -300,34 +800,130 @@ def _post(payload: dict[str, Any], url: str) -> bool:
|
||||
|
||||
|
||||
def resolve_distinct_id() -> tuple[str, str]:
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any."""
|
||||
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
||||
|
||||
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
||||
anonymous id: aliasing one account email to another merges two real person
|
||||
profiles and cannot be undone, so a key that now belongs to a different
|
||||
account re-resolves with no alias.
|
||||
"""
|
||||
identity = _read_identity()
|
||||
email = identity.get("email", "")
|
||||
if email:
|
||||
return email, ""
|
||||
key = memory_core.api_key()
|
||||
fingerprint = _digest(key) if key else ""
|
||||
email = identity.get("email", "")
|
||||
|
||||
if email and fingerprint:
|
||||
recorded = identity.get("key_fingerprint", "")
|
||||
if recorded == fingerprint:
|
||||
return email, ""
|
||||
if not recorded:
|
||||
# Rows written before fingerprints existed. Verify rather than
|
||||
# adopt: a key changed before the upgrade would otherwise bind the
|
||||
# new key to the previous account's email, permanently, and the
|
||||
# fingerprint would then agree with itself forever after.
|
||||
verified = _resolve_email(key)
|
||||
if not verified:
|
||||
# Offline, firewalled, or the API is down. Keep the previous
|
||||
# behaviour and retry on the next flush rather than dropping a
|
||||
# real account attribution. Safe because the same network that
|
||||
# failed /v1/ping/ is about to fail the PostHog POST, so nothing
|
||||
# is delivered under the unverified identity in the meantime.
|
||||
return email, ""
|
||||
identity["email"] = verified
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return verified, ""
|
||||
|
||||
if not key:
|
||||
# No key to verify the account with; do not keep attributing to it.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
email = _resolve_email(key)
|
||||
if not email:
|
||||
|
||||
resolved = _resolve_email(key)
|
||||
if not resolved:
|
||||
# The key changed and will not resolve (revoked, offline, API down).
|
||||
# Reaching here with an email means the recorded fingerprint disagreed,
|
||||
# so the key really did change. Drop the account and rotate: the stored
|
||||
# anonymous id may already be merged into that account's person, and
|
||||
# reusing it would keep the events on the profile we are trying to
|
||||
# leave.
|
||||
if email:
|
||||
identity.pop("email", None)
|
||||
identity.pop("key_fingerprint", None)
|
||||
return _rotate_anonymous_id(identity), ""
|
||||
return anonymous_id(identity), ""
|
||||
previous = identity.get("anonymous_id", "")
|
||||
identity["email"] = email
|
||||
|
||||
# Alias only when going anonymous -> email for the first time. Once an anon
|
||||
# id has been merged into an account it must never be offered again: an
|
||||
# alias naming an already-identified id is what could link two real people.
|
||||
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
||||
if previous:
|
||||
identity["aliased"] = True
|
||||
identity["email"] = resolved
|
||||
identity["key_fingerprint"] = fingerprint
|
||||
_write_identity(identity)
|
||||
return email, previous
|
||||
return resolved, previous
|
||||
|
||||
|
||||
def flush() -> int:
|
||||
"""Drain claimed spools to PostHog and return the number of events sent."""
|
||||
"""Drain the live spool, then any parked claims, and return events sent."""
|
||||
if not is_enabled():
|
||||
return 0
|
||||
claim = _claim_spool()
|
||||
sent, delivered = _drain(_claim_spool())
|
||||
if not delivered:
|
||||
# The network is failing. Retrying other batches now would only burn
|
||||
# their attempt budget against the same broken connection.
|
||||
return sent
|
||||
|
||||
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
||||
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
||||
directory = memory_core.data_dir()
|
||||
_sweep_debris(directory)
|
||||
for _ in range(MAX_PARKED_PER_RUN):
|
||||
parked = _claim_parked(directory)
|
||||
if parked is None:
|
||||
break
|
||||
count, delivered = _drain(parked)
|
||||
sent += count
|
||||
if not delivered:
|
||||
break
|
||||
return sent
|
||||
|
||||
|
||||
def _drain(claim: Path | None) -> tuple[int, bool]:
|
||||
"""Post one claimed batch file, recording progress after every batch.
|
||||
|
||||
Returns (events sent, whether everything was delivered).
|
||||
"""
|
||||
if claim is None:
|
||||
return 0
|
||||
return 0, True
|
||||
try:
|
||||
lines = claim.read_text(encoding="utf-8").splitlines()
|
||||
except ValueError:
|
||||
# UnicodeDecodeError from a torn write: the content is unrecoverable, so
|
||||
# quarantine rather than retry. flush() runs from a bare `finally:` in
|
||||
# flush_worker, so raising here also skips the handoff cleanup, and an
|
||||
# undecodable file would otherwise be re-read on every flush forever.
|
||||
# Reported as delivered because there is nothing left to deliver and the
|
||||
# rest of the run should continue.
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt"))
|
||||
except OSError:
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0, True
|
||||
except OSError:
|
||||
return 0
|
||||
# Could not read it, which is not the same as having nothing to send.
|
||||
# The file is left exactly where it is: a vanished or briefly unreadable
|
||||
# claim is retryable, and quarantining it here would discard events over
|
||||
# a transient filesystem error. Reported as undelivered so the run stops
|
||||
# instead of counting a batch nothing was posted from as delivered.
|
||||
return 0, False
|
||||
events = []
|
||||
for line in lines:
|
||||
try:
|
||||
@@ -337,11 +933,18 @@ def flush() -> int:
|
||||
if isinstance(value, dict) and value.get("event"):
|
||||
events.append(value)
|
||||
if not events:
|
||||
# Only delete when the file really is empty. A non-empty file that
|
||||
# parses to nothing is a torn write, and its contents are the unsent
|
||||
# remainder — deleting it is the data loss this PR exists to prevent.
|
||||
try:
|
||||
claim.unlink()
|
||||
empty = claim.stat().st_size == 0
|
||||
except OSError:
|
||||
empty = True
|
||||
try:
|
||||
claim.replace(claim.with_suffix(".corrupt")) if not empty else claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
return 0, True
|
||||
|
||||
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
||||
if aliased_anonymous_id:
|
||||
@@ -360,12 +963,17 @@ def flush() -> int:
|
||||
|
||||
sent = 0
|
||||
for start in range(0, len(events), BATCH_SIZE):
|
||||
chunk = events[start : start + BATCH_SIZE]
|
||||
batch = [
|
||||
{
|
||||
"event": event["event"],
|
||||
"distinct_id": distinct_id,
|
||||
# Carried through from record() so a resend can be collapsed.
|
||||
"uuid": event.get("uuid"),
|
||||
"timestamp": event.get("timestamp"),
|
||||
"properties": {
|
||||
# Fallback only: events recorded by a build before source
|
||||
# moved into record() have none of their own.
|
||||
"source": _source_tag,
|
||||
"language": "python",
|
||||
"$process_person_profile": False,
|
||||
@@ -373,16 +981,24 @@ def flush() -> int:
|
||||
**(event.get("properties") or {}),
|
||||
},
|
||||
}
|
||||
for event in events[start : start + BATCH_SIZE]
|
||||
for event in chunk
|
||||
]
|
||||
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
||||
return sent
|
||||
sent += len(batch)
|
||||
try:
|
||||
claim.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
return sent
|
||||
# Keep only what has not been delivered, and release the lease.
|
||||
# Previously the whole file was kept and the retry re-posted every
|
||||
# batch, including the ones that had already arrived.
|
||||
_release_claim(claim, events[start:])
|
||||
return sent, False
|
||||
sent += len(chunk)
|
||||
# Record progress and refresh the lease after each successful batch, so
|
||||
# a crash repeats at most one batch instead of the entire file. If the
|
||||
# rewrite fails the claim still holds delivered events, so stop rather
|
||||
# than carry on as though progress were recorded — continuing is how the
|
||||
# duplicate delivery this PR fixes would come back.
|
||||
if not _rewrite_claim(claim, events[start + len(chunk) :]):
|
||||
_release_claim(claim, events[start + len(chunk) :])
|
||||
return sent, False
|
||||
return sent, True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json",
|
||||
"name": "mem0",
|
||||
"version": "0.3.1",
|
||||
"version": "0.3.2",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"author": {
|
||||
"name": "Mem0",
|
||||
|
||||
@@ -6,8 +6,8 @@ description: Pause Mem0 memory capture on this machine. Use when the user wants
|
||||
# Pause memory capture
|
||||
|
||||
To pause (hooks stop capturing and sending session content; a minimal
|
||||
anonymous telemetry ping still fires at session start unless
|
||||
`MEM0_TELEMETRY=false`):
|
||||
telemetry ping still fires at session start, under your Mem0 account email,
|
||||
unless `MEM0_TELEMETRY=false`):
|
||||
|
||||
```bash
|
||||
python3 "${PLUGIN_ROOT}/core/memory_cli.py" --harness "coding-agent" pause
|
||||
|
||||
@@ -79,9 +79,11 @@ the tool share one Mem0 backend and namespace.
|
||||
|
||||
## Telemetry
|
||||
|
||||
The store sends anonymous usage events (store configuration, operation, duration,
|
||||
result counts, coarse failure kind) over the Mem0 SDK's existing telemetry client,
|
||||
tagged `source="STRANDS"`. Queries, memory text, message content, entity ids, and
|
||||
The store sends usage events (store configuration, operation, duration, result
|
||||
counts, coarse failure kind) over the Mem0 SDK's existing telemetry client,
|
||||
tagged `source="STRANDS"`. These are **not anonymous**: when an API key is
|
||||
configured they are sent under your Mem0 account email, the same way the SDK
|
||||
attributes its own. Queries, memory text, message content, entity ids, and
|
||||
metadata are never sent. Turn it off with `MEM0_TELEMETRY=false`.
|
||||
|
||||
## Development
|
||||
|
||||
@@ -35,6 +35,8 @@ export interface PluginAuthConfig {
|
||||
autoCapture?: boolean;
|
||||
topK?: number;
|
||||
anonymousTelemetryId?: string;
|
||||
/** SHA-256 prefix of the API key userEmail was resolved for. */
|
||||
keyFingerprint?: string;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -135,6 +137,9 @@ export function readPluginAuth(): PluginAuthConfig {
|
||||
autoCapture: cfg.autoCapture as boolean | undefined,
|
||||
topK: cfg.topK as number | undefined,
|
||||
anonymousTelemetryId: cfg.anonymousTelemetryId as string | undefined,
|
||||
// Without this the reader silently drops it, every fingerprint comparison
|
||||
// fails against undefined, and the resolved email is never used again.
|
||||
keyFingerprint: cfg.keyFingerprint as string | undefined,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -236,6 +241,21 @@ export function getBaseUrl(): string {
|
||||
return auth.baseUrl || DEFAULT_BASE_URL;
|
||||
}
|
||||
|
||||
/** Forget the resolved account, so the next capture re-resolves for the current key. */
|
||||
export function clearResolvedAccount(): void {
|
||||
const full = readFullConfig() as any;
|
||||
const cfg = full?.plugins?.entries?.[PLUGIN_ID]?.config;
|
||||
if (!cfg) return;
|
||||
let changed = false;
|
||||
for (const key of ["userEmail", "keyFingerprint"]) {
|
||||
if (key in cfg) {
|
||||
delete cfg[key];
|
||||
changed = true;
|
||||
}
|
||||
}
|
||||
if (changed) writeFullConfig(full);
|
||||
}
|
||||
|
||||
/** Remove anonymousTelemetryId from config (after PostHog aliasing) */
|
||||
export function clearAnonymousTelemetryId(): void {
|
||||
const full = readFullConfig() as any;
|
||||
|
||||
@@ -148,6 +148,7 @@ const ALLOWED_KEYS = [
|
||||
"mode",
|
||||
"apiKey",
|
||||
"anonymousTelemetryId",
|
||||
"keyFingerprint",
|
||||
"baseUrl",
|
||||
"userId",
|
||||
"userEmail",
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"id": "openclaw-mem0",
|
||||
"name": "Memory (Mem0)",
|
||||
"description": "Mem0 memory backend for OpenClaw — platform (mem0.ai cloud) or self-hosted open-source. Auto-recall and auto-capture are opt-in (disabled by default). Supports OpenAI, Anthropic, Ollama (fully local), Qdrant, and PGVector providers.",
|
||||
"version": "1.1.0",
|
||||
"version": "1.2.0",
|
||||
"kind": "memory",
|
||||
"skills": ["skills"],
|
||||
"commandAliases": [
|
||||
@@ -209,6 +209,10 @@
|
||||
"type": "string",
|
||||
"description": "Persistent anonymous telemetry identifier"
|
||||
},
|
||||
"keyFingerprint": {
|
||||
"type": "string",
|
||||
"description": "Digest of the API key userEmail was resolved for. Set automatically; a mismatch re-resolves the account."
|
||||
},
|
||||
"oss": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@mem0/openclaw-mem0",
|
||||
"version": "1.1.0",
|
||||
"version": "1.2.0",
|
||||
"type": "module",
|
||||
"description": "Mem0 memory backend for OpenClaw — platform or self-hosted open-source",
|
||||
"license": "Apache-2.0",
|
||||
|
||||
@@ -1,14 +1,20 @@
|
||||
import { createHash, randomUUID } from "node:crypto";
|
||||
|
||||
import { createTelemetry } from "../agent-plugin-core/typescript/src/telemetry.ts";
|
||||
import { clearAnonymousTelemetryId, getBaseUrl, readPluginAuth, writePluginAuth } from "./cli/config-file.ts";
|
||||
import {
|
||||
clearAnonymousTelemetryId,
|
||||
clearResolvedAccount,
|
||||
getBaseUrl,
|
||||
readPluginAuth,
|
||||
writePluginAuth,
|
||||
} from "./cli/config-file.ts";
|
||||
|
||||
declare const __OPENCLAW_PLUGIN_VERSION__: string;
|
||||
export const PLUGIN_VERSION: string = __OPENCLAW_PLUGIN_VERSION__;
|
||||
|
||||
let cachedAnonymousId: string | undefined;
|
||||
let aliasCheckDone = false;
|
||||
let emailResolutionAttempted = false;
|
||||
let resolutionAttemptedFor = "";
|
||||
let currentDistinctId = "";
|
||||
|
||||
function enabled(): boolean {
|
||||
@@ -33,10 +39,35 @@ function anonymousId(): string {
|
||||
return (cachedAnonymousId = created);
|
||||
}
|
||||
|
||||
/** SHA-256 prefix of the key an account was resolved for. */
|
||||
function keyFingerprint(apiKey?: string): string {
|
||||
return apiKey ? createHash("sha256").update(apiKey).digest("hex").slice(0, 16) : "";
|
||||
}
|
||||
|
||||
function distinctId(apiKey?: string): string {
|
||||
try {
|
||||
const email = readPluginAuth().userEmail;
|
||||
if (email) return createHash("sha256").update(email).digest("hex");
|
||||
const auth = readPluginAuth();
|
||||
if (auth.userEmail) {
|
||||
// Only when it belongs to the key in hand. Without this check a cached
|
||||
// email was used forever: switch to a different account and every event
|
||||
// kept reporting under the previous one, with nothing to notice it by.
|
||||
if (auth.keyFingerprint === keyFingerprint(apiKey)) {
|
||||
return createHash("sha256").update(auth.userEmail).digest("hex");
|
||||
}
|
||||
// Only a REAL key that disagrees means the account changed. Without the
|
||||
// apiKey guard the comparison is `undefined === ""` for any call that
|
||||
// simply omits the key, so a capture with no context wiped a perfectly
|
||||
// good account out of openclaw.json.
|
||||
//
|
||||
// A row with an email and NO fingerprint is the legacy shape, from an
|
||||
// install predating this field. Clearing it here deleted a real account
|
||||
// before anything had replaced it, and if the re-resolve then failed
|
||||
// because the user was offline the email was gone from disk for good. The
|
||||
// Python core refuses the same trade: verify, and keep what you have until
|
||||
// the verification succeeds. resolveEmail below overwrites both fields
|
||||
// when it does, so there is nothing to clear first.
|
||||
if (apiKey && auth.keyFingerprint) clearResolvedAccount();
|
||||
}
|
||||
} catch {
|
||||
// Fall through to the API key or anonymous identity.
|
||||
}
|
||||
@@ -66,8 +97,16 @@ function identifyAnonymous(id: string): void {
|
||||
}
|
||||
|
||||
function resolveEmail(apiKey: string): void {
|
||||
if (emailResolutionAttempted) return;
|
||||
emailResolutionAttempted = true;
|
||||
// Latched per key, not once per process. A single boolean meant a key changed
|
||||
// mid-session was never looked up, so the fallback identity stuck until restart.
|
||||
const fingerprint = keyFingerprint(apiKey);
|
||||
if (resolutionAttemptedFor === fingerprint) return;
|
||||
resolutionAttemptedFor = fingerprint;
|
||||
const releaseLatch = () => {
|
||||
// A failed lookup must not pin the fallback identity for the rest of the
|
||||
// process. Released so the next capture tries again.
|
||||
if (resolutionAttemptedFor === fingerprint) resolutionAttemptedFor = "";
|
||||
};
|
||||
fetch(`${getBaseUrl().replace(/\/+$/, "")}/v1/ping/`, {
|
||||
method: "GET",
|
||||
headers: { Authorization: `Token ${apiKey}`, "Content-Type": "application/json" },
|
||||
@@ -76,7 +115,7 @@ function resolveEmail(apiKey: string): void {
|
||||
.then((response) => response.json())
|
||||
.then((data: any) => {
|
||||
if (!data?.user_email) return;
|
||||
writePluginAuth({ userEmail: data.user_email });
|
||||
writePluginAuth({ userEmail: data.user_email, keyFingerprint: fingerprint });
|
||||
const oldId = createHash("sha256").update(apiKey).digest("hex");
|
||||
const newId = createHash("sha256").update(data.user_email).digest("hex");
|
||||
for (const event of telemetry.queueForTesting()) {
|
||||
@@ -84,7 +123,8 @@ function resolveEmail(apiKey: string): void {
|
||||
}
|
||||
})
|
||||
.catch(() => {
|
||||
// The API-key hash remains a stable fallback.
|
||||
// The API-key hash remains a stable fallback, and the next capture retries.
|
||||
releaseLatch();
|
||||
});
|
||||
}
|
||||
|
||||
@@ -96,13 +136,15 @@ export function captureEvent(
|
||||
if (!enabled()) return;
|
||||
try {
|
||||
currentDistinctId = distinctId(context?.apiKey);
|
||||
let hasEmail = false;
|
||||
let resolvedForThisKey = false;
|
||||
try {
|
||||
hasEmail = Boolean(readPluginAuth().userEmail);
|
||||
const auth = readPluginAuth();
|
||||
resolvedForThisKey =
|
||||
Boolean(auth.userEmail) && auth.keyFingerprint === keyFingerprint(context?.apiKey);
|
||||
} catch {
|
||||
// Resolve it below when possible.
|
||||
}
|
||||
if (context?.apiKey && !hasEmail) resolveEmail(context.apiKey);
|
||||
if (context?.apiKey && !resolvedForThisKey) resolveEmail(context.apiKey);
|
||||
identifyAnonymous(currentDistinctId);
|
||||
telemetry.capture(eventName, {
|
||||
mode: context?.mode,
|
||||
|
||||
@@ -237,3 +237,43 @@ describe("getBaseUrl", () => {
|
||||
expect(getBaseUrl()).toBe(DEFAULT_BASE_URL);
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// keyFingerprint round trip
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe("keyFingerprint survives a write and read", () => {
|
||||
it("readPluginAuth returns a persisted keyFingerprint", () => {
|
||||
// It did not. readPluginAuth builds its result field by field, and this one
|
||||
// was missing, so every fingerprint comparison ran against undefined, the
|
||||
// resolved email was never used again, and telemetry silently fell back to
|
||||
// the API key hash. The telemetry tests could not see it because they mock
|
||||
// this module and their mock returned the field the real reader dropped.
|
||||
setConfigFile({
|
||||
plugins: {
|
||||
entries: {
|
||||
"openclaw-mem0": {
|
||||
config: {
|
||||
apiKey: "m0-test",
|
||||
userEmail: "person@example.com",
|
||||
keyFingerprint: "0123456789abcdef",
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(readPluginAuth().keyFingerprint).toBe("0123456789abcdef");
|
||||
});
|
||||
|
||||
it("writePluginAuth persists it where readPluginAuth looks", () => {
|
||||
setConfigFile({ plugins: { entries: { "openclaw-mem0": { config: {} } } } });
|
||||
|
||||
writePluginAuth({ userEmail: "person@example.com", keyFingerprint: "abc123" });
|
||||
|
||||
const written = JSON.parse(mockWriteText.mock.calls.at(-1)![1] as string);
|
||||
const cfg = written.plugins.entries["openclaw-mem0"].config;
|
||||
expect(cfg.keyFingerprint).toBe("abc123");
|
||||
expect(cfg.userEmail).toBe("person@example.com");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -485,3 +485,18 @@ describe("mem0ConfigSchema.parse() — apiKey edge cases", () => {
|
||||
expect(cfg.needsSetup).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("telemetry fingerprint round trip", () => {
|
||||
it("a config carrying keyFingerprint is accepted by the real schema", () => {
|
||||
// What writePluginAuth persists after a successful lookup. It was not in
|
||||
// ALLOWED_KEYS, and assertAllowedKeys throws, so the first successful
|
||||
// resolve wrote a config that broke every subsequent load of the plugin.
|
||||
const persisted = {
|
||||
apiKey: "m0-test",
|
||||
userEmail: "person@example.com",
|
||||
keyFingerprint: "0123456789abcdef",
|
||||
};
|
||||
|
||||
expect(() => mem0ConfigSchema.parse(persisted)).not.toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -3,15 +3,30 @@ import { describe, it, expect, vi, beforeEach, afterEach } from "vitest";
|
||||
// Mock config-file before importing telemetry
|
||||
vi.mock("../cli/config-file.ts", () => ({
|
||||
readPluginAuth: vi.fn().mockReturnValue({}),
|
||||
writePluginAuth: vi.fn(),
|
||||
clearAnonymousTelemetryId: vi.fn(),
|
||||
clearResolvedAccount: vi.fn(),
|
||||
getBaseUrl: vi.fn().mockReturnValue("https://api.mem0.ai"),
|
||||
}));
|
||||
|
||||
import { captureEvent } from "../telemetry.ts";
|
||||
import { readPluginAuth } from "../cli/config-file.ts";
|
||||
import { clearResolvedAccount, readPluginAuth } from "../cli/config-file.ts";
|
||||
|
||||
/** sha256(key).slice(0, 16), the shape telemetry.ts stores. */
|
||||
async function fingerprintOf(apiKey: string): Promise<string> {
|
||||
const { createHash } = await import("node:crypto");
|
||||
return createHash("sha256").update(apiKey).digest("hex").slice(0, 16);
|
||||
}
|
||||
|
||||
describe("telemetry", () => {
|
||||
let fetchSpy: ReturnType<typeof vi.fn>;
|
||||
|
||||
beforeEach(() => {
|
||||
// Call history has to be cleared per test, not just restored: the mocks are
|
||||
// module-level vi.fn()s, so without this one test's calls are visible to the
|
||||
// next and assertions on "was not called" pass or fail by ordering.
|
||||
vi.clearAllMocks();
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockReturnValue({});
|
||||
// Reset telemetry enabled state
|
||||
(globalThis as any).__mem0_telemetry_override = undefined;
|
||||
fetchSpy = vi.fn().mockResolvedValue({ ok: true });
|
||||
@@ -40,11 +55,31 @@ describe("telemetry", () => {
|
||||
expect(() => captureEvent("test_event")).not.toThrow();
|
||||
});
|
||||
|
||||
it("uses userEmail as distinct ID when available", () => {
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockReturnValueOnce({
|
||||
it("uses userEmail as distinct ID when it belongs to the current key", async () => {
|
||||
// Previously asserted only not.toThrow(), which passed whatever the identity
|
||||
// turned out to be, and under the fingerprint gate the no-context path does
|
||||
// not use the email at all. Pin the real condition instead.
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockReturnValue({
|
||||
userEmail: "test@example.com",
|
||||
keyFingerprint: await fingerprintOf("key-a"),
|
||||
});
|
||||
expect(() => captureEvent("test_event")).not.toThrow();
|
||||
|
||||
captureEvent("test_event", {}, { apiKey: "key-a" });
|
||||
|
||||
expect(clearResolvedAccount).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("a capture with no apiKey leaves a resolved account alone", async () => {
|
||||
// `undefined === ""` made every keyless capture look like a key change, so
|
||||
// one context-free call wiped a good account out of openclaw.json.
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockReturnValue({
|
||||
userEmail: "test@example.com",
|
||||
keyFingerprint: await fingerprintOf("key-a"),
|
||||
});
|
||||
|
||||
captureEvent("test_event");
|
||||
|
||||
expect(clearResolvedAccount).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("falls back to a generated anonymous id when no apiKey", () => {
|
||||
@@ -52,6 +87,43 @@ describe("telemetry", () => {
|
||||
expect(() => captureEvent("test_event", {}, {})).not.toThrow();
|
||||
});
|
||||
|
||||
it("keeps using a cached email only while it belongs to the current key", async () => {
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockReturnValue({
|
||||
userEmail: "person@example.com",
|
||||
keyFingerprint: await fingerprintOf("key-a"),
|
||||
});
|
||||
|
||||
captureEvent("test_event", {}, { apiKey: "key-a" });
|
||||
|
||||
expect(clearResolvedAccount).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("forgets the account when the API key changes", async () => {
|
||||
// The defect: the cached email was used forever, so events after an account
|
||||
// switch kept reporting under the previous account.
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockReturnValue({
|
||||
userEmail: "person@example.com",
|
||||
keyFingerprint: await fingerprintOf("key-a"),
|
||||
});
|
||||
|
||||
captureEvent("test_event", {}, { apiKey: "key-b" });
|
||||
|
||||
expect(clearResolvedAccount).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("re-resolves for a key it has not looked up before", async () => {
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockReturnValue({
|
||||
userEmail: "person@example.com",
|
||||
keyFingerprint: await fingerprintOf("key-a"),
|
||||
});
|
||||
|
||||
captureEvent("test_event", {}, { apiKey: "key-c" });
|
||||
|
||||
// The resolution latch is per key, not once per process, so a key changed
|
||||
// mid-session is actually looked up instead of sticking to the fallback.
|
||||
expect(fetchSpy).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("handles readPluginAuth errors gracefully", () => {
|
||||
(readPluginAuth as ReturnType<typeof vi.fn>).mockImplementationOnce(() => {
|
||||
throw new Error("config read failed");
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@mem0/opencode-plugin",
|
||||
"version": "0.3.0",
|
||||
"version": "0.4.0",
|
||||
"type": "module",
|
||||
"description": "Mem0 persistent memory plugin for OpenCode — add, search, and manage memories across sessions",
|
||||
"main": "dist/index.js",
|
||||
@@ -38,6 +38,7 @@
|
||||
"build": "bun build opencode-mem0.ts --outdir dist --target bun --format esm --entry-naming index.[ext]",
|
||||
"dev": "bun build opencode-mem0.ts --outdir dist --target bun --format esm --entry-naming index.[ext] --watch",
|
||||
"type-check": "tsc --noEmit",
|
||||
"test": "bun test",
|
||||
"prepack": "bun run build",
|
||||
"postpack": ""
|
||||
},
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import { createHash } from "node:crypto";
|
||||
|
||||
import { afterEach, describe, expect, test } from "bun:test";
|
||||
import { buildEvent, captureEvent, isTelemetryEnabled } from "./telemetry";
|
||||
|
||||
@@ -13,7 +15,10 @@ describe("opencode telemetry", () => {
|
||||
expect(payload).not.toBeNull();
|
||||
const props = payload!.properties as Record<string, unknown>;
|
||||
expect(payload!.event).toBe("plugin.session_start");
|
||||
expect(props.source).toBe("plugin");
|
||||
// Was "plugin", which named no particular plugin and matched no vocabulary.
|
||||
// Now shaped like every other surface. Any saved PostHog insight filtering
|
||||
// source = "plugin" needs repointing; historical data is untouched.
|
||||
expect(props.source).toBe("OPENCODE_PLUGIN");
|
||||
expect(props.platform).toBe("opencode");
|
||||
expect(props.memory_count).toBe(5);
|
||||
expect(props.$process_person_profile).toBe(false);
|
||||
@@ -30,7 +35,7 @@ describe("opencode telemetry", () => {
|
||||
const props = buildEvent("x", { platform: "HACK", source: "HACK" }, KEY)!
|
||||
.properties as Record<string, unknown>;
|
||||
expect(props.platform).toBe("opencode");
|
||||
expect(props.source).toBe("plugin");
|
||||
expect(props.source).toBe("OPENCODE_PLUGIN");
|
||||
});
|
||||
|
||||
test("returns null without an API key (no anonymous events)", () => {
|
||||
@@ -56,9 +61,12 @@ describe("opencode telemetry", () => {
|
||||
expect(typeof props.os_version).toBe("string");
|
||||
});
|
||||
|
||||
test("project_hash is sha256(projectId) when a project id is supplied", async () => {
|
||||
test("project_hash is a salted digest of the project id", async () => {
|
||||
// Previously asserted the bare sha256(projectId), which is the defect: that
|
||||
// digest is reversible by anyone who can guess a project id. Salted with the
|
||||
// API key, which is already in play here and is high entropy.
|
||||
const { createHash } = await import("node:crypto");
|
||||
const expected = createHash("sha256").update("acme-repo").digest("hex");
|
||||
const expected = createHash("sha256").update(`${KEY}:acme-repo`).digest("hex");
|
||||
const props = buildEvent("session_start", {}, KEY, "acme-repo")!
|
||||
.properties as Record<string, unknown>;
|
||||
expect(props.project_hash).toBe(expected);
|
||||
@@ -76,3 +84,38 @@ describe("opencode telemetry", () => {
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe("project_hash salting", () => {
|
||||
const PROJECT = "my-project";
|
||||
|
||||
test("is not a bare digest of the project id", () => {
|
||||
// The defect: an unsalted SHA-256 over a guessable identifier is reversible
|
||||
// by anyone who can enumerate project ids.
|
||||
const unsalted = createHash("sha256").update(PROJECT).digest("hex");
|
||||
const payload = buildEvent("session_start", {}, KEY, PROJECT) as Record<string, any>;
|
||||
|
||||
expect(payload.properties.project_hash).toBeDefined();
|
||||
expect(payload.properties.project_hash).not.toBe(unsalted);
|
||||
});
|
||||
|
||||
test("differs per account for the same project", () => {
|
||||
const a = buildEvent("session_start", {}, "m0-account-a", PROJECT) as Record<string, any>;
|
||||
const b = buildEvent("session_start", {}, "m0-account-b", PROJECT) as Record<string, any>;
|
||||
|
||||
expect(a.properties.project_hash).not.toBe(b.properties.project_hash);
|
||||
});
|
||||
|
||||
test("is stable for one account, so joins still work", () => {
|
||||
const first = buildEvent("session_start", {}, KEY, PROJECT) as Record<string, any>;
|
||||
const second = buildEvent("session_end", {}, KEY, PROJECT) as Record<string, any>;
|
||||
|
||||
expect(first.properties.project_hash).toBe(second.properties.project_hash);
|
||||
});
|
||||
|
||||
test("is omitted rather than unsalted when there is no key", () => {
|
||||
const payload = buildEvent("session_start", {}, undefined, PROJECT);
|
||||
|
||||
// No key means no event at all, so there is no unsalted hash to leak.
|
||||
expect(payload).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -25,7 +25,9 @@ const PLUGIN_VERSION = (() => {
|
||||
let currentDistinctId = "";
|
||||
const telemetry = createTelemetry({
|
||||
host: "opencode",
|
||||
source: "plugin",
|
||||
// Shaped like the platform's EventSource values, as every other surface is.
|
||||
// "plugin" said nothing about which plugin and matched no vocabulary.
|
||||
source: "OPENCODE_PLUGIN",
|
||||
version: PLUGIN_VERSION,
|
||||
distinctId: () => currentDistinctId,
|
||||
eventName: (event) => `plugin.${event}`,
|
||||
@@ -40,8 +42,23 @@ function distinctId(apiKey: string): string {
|
||||
return createHash("sha256").update(apiKey).digest("hex").slice(0, 32);
|
||||
}
|
||||
|
||||
function projectHash(projectId?: string): Record<string, string> {
|
||||
return projectId ? { project_hash: createHash("sha256").update(projectId).digest("hex") } : {};
|
||||
/**
|
||||
* Salted so the hash is not enumerable.
|
||||
*
|
||||
* An unsalted SHA-256 of a project id is reversible by anyone who can guess the
|
||||
* id, which for a project identifier is a small space. The API key is the salt:
|
||||
* it is already in play here (distinctId is a digest of it), it is high entropy,
|
||||
* and using it needs no per-install file and so no write race to get wrong. The
|
||||
* hash is therefore per account rather than per machine, which also keeps joins
|
||||
* working for one user across machines. It resets when the key rotates, which is
|
||||
* consistent, because distinctId resets with it.
|
||||
*
|
||||
* Both are omitted without a key. An event cannot be built without a distinctId
|
||||
* anyway, so this costs nothing.
|
||||
*/
|
||||
function projectHash(projectId?: string, apiKey?: string): Record<string, string> {
|
||||
if (!projectId || !apiKey) return {};
|
||||
return { project_hash: createHash("sha256").update(`${apiKey}:${projectId}`).digest("hex") };
|
||||
}
|
||||
|
||||
export function buildEvent(
|
||||
@@ -51,7 +68,7 @@ export function buildEvent(
|
||||
projectId?: string,
|
||||
): Record<string, unknown> | null {
|
||||
currentDistinctId = apiKey ? distinctId(apiKey) : "";
|
||||
const event = telemetry.build(eventType, { ...properties, ...projectHash(projectId) });
|
||||
const event = telemetry.build(eventType, { ...properties, ...projectHash(projectId, apiKey) });
|
||||
return event ? { api_key: POSTHOG_API_KEY, ...event } : null;
|
||||
}
|
||||
|
||||
@@ -62,5 +79,5 @@ export function captureEvent(
|
||||
projectId?: string,
|
||||
): void {
|
||||
currentDistinctId = apiKey ? distinctId(apiKey) : "";
|
||||
telemetry.capture(eventType, { ...properties, ...projectHash(projectId) });
|
||||
telemetry.capture(eventType, { ...properties, ...projectHash(projectId, apiKey) });
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@mem0/pi-agent-plugin",
|
||||
"version": "0.3.0",
|
||||
"version": "0.3.1",
|
||||
"type": "module",
|
||||
"description": "Mem0 memory extension for Pi Agent persistent, scoped, semantic memory across sessions and projects",
|
||||
"license": "Apache-2.0",
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { applySurfaceHeaders, PLATFORM_APPLICATION, PLATFORM_SOURCE } from "./attribution.ts";
|
||||
|
||||
function client(headers: Record<string, string> = {}) {
|
||||
return { headers: { Authorization: "Token k", ...headers } } as never;
|
||||
}
|
||||
|
||||
describe("applySurfaceHeaders", () => {
|
||||
it("stamps the shared client so every path is attributed, not just commands", () => {
|
||||
const mem0 = client();
|
||||
applySurfaceHeaders(mem0);
|
||||
const headers = (mem0 as unknown as { headers: Record<string, string> }).headers;
|
||||
|
||||
expect(headers["X-Mem0-Source"]).toBe(PLATFORM_SOURCE);
|
||||
expect(headers["X-Application"]).toBe(PLATFORM_APPLICATION);
|
||||
expect(headers["X-Mem0-Client"]).toMatch(/^mem0-pi-agent\//);
|
||||
expect(headers.Authorization).toBe("Token k");
|
||||
});
|
||||
|
||||
it("defers to a surface an outer wrapper already declared", () => {
|
||||
const mem0 = client({ "X-Mem0-Source": "OPENCLAW", "X-Application": "vscode" });
|
||||
applySurfaceHeaders(mem0);
|
||||
const headers = (mem0 as unknown as { headers: Record<string, string> }).headers;
|
||||
|
||||
expect(headers["X-Mem0-Source"]).toBe("OPENCLAW");
|
||||
expect(headers["X-Application"]).toBe("vscode");
|
||||
});
|
||||
|
||||
it("appends to the client stack rather than replacing it", () => {
|
||||
const mem0 = client({ "X-Mem0-Client": "openclaw/2.1.0" });
|
||||
applySurfaceHeaders(mem0);
|
||||
const headers = (mem0 as unknown as { headers: Record<string, string> }).headers;
|
||||
|
||||
expect(headers["X-Mem0-Client"]).toMatch(/^openclaw\/2\.1\.0, mem0-pi-agent\//);
|
||||
});
|
||||
|
||||
it("treats a blank header as absent", () => {
|
||||
const mem0 = client({ "X-Mem0-Source": " " });
|
||||
applySurfaceHeaders(mem0);
|
||||
const headers = (mem0 as unknown as { headers: Record<string, string> }).headers;
|
||||
|
||||
expect(headers["X-Mem0-Source"]).toBe(PLATFORM_SOURCE);
|
||||
});
|
||||
|
||||
it("bounds the stack so a long chain cannot grow the header without limit", () => {
|
||||
const mem0 = client({ "X-Mem0-Client": "a/1, b/1, c/1, d/1, e/1" });
|
||||
applySurfaceHeaders(mem0);
|
||||
const headers = (mem0 as unknown as { headers: Record<string, string> }).headers;
|
||||
|
||||
expect(headers["X-Mem0-Client"].split(",").length).toBeLessThanOrEqual(4);
|
||||
expect(headers["X-Mem0-Client"].length).toBeLessThanOrEqual(200);
|
||||
});
|
||||
});
|
||||
|
||||
describe("client stack bounding", () => {
|
||||
it("keeps our own entry when the caller already filled the stack", () => {
|
||||
// The defect: pushing then trimming to four dropped exactly the entry this
|
||||
// function exists to add, so we vanished from our own stack.
|
||||
const mem0 = client({ "X-Mem0-Client": "a/1, b/2, c/3, d/4" });
|
||||
applySurfaceHeaders(mem0);
|
||||
const stack = (mem0 as unknown as { headers: Record<string, string> }).headers["X-Mem0-Client"];
|
||||
|
||||
expect(stack).toMatch(/mem0-pi-agent\//);
|
||||
expect(stack.split(",").length).toBeLessThanOrEqual(4);
|
||||
});
|
||||
|
||||
it("drops whole entries at the character cap, never a fragment", () => {
|
||||
const long = `${"n".repeat(90)}/1.0, ${"m".repeat(90)}/1.0, ${"o".repeat(90)}/1.0`;
|
||||
const mem0 = client({ "X-Mem0-Client": long });
|
||||
applySurfaceHeaders(mem0);
|
||||
const stack = (mem0 as unknown as { headers: Record<string, string> }).headers["X-Mem0-Client"];
|
||||
|
||||
expect(stack.length).toBeLessThanOrEqual(200);
|
||||
expect(stack.endsWith("/0.0.0") || /mem0-pi-agent\/[\w.\-]+$/.test(stack)).toBe(true);
|
||||
// Every surviving entry is whole: name/version, no severed tail.
|
||||
for (const entry of stack.split(",")) {
|
||||
expect(entry.trim()).toMatch(/^[^/]+\/[^/]+$/);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,65 @@
|
||||
import type MemoryClient from "mem0ai";
|
||||
import * as fs from "node:fs";
|
||||
|
||||
/** Surface identity for this plugin, as the platform's EventSource knows it. */
|
||||
export const PLATFORM_SOURCE = "PI_AGENT";
|
||||
|
||||
/** Host app the plugin runs inside. Allowlisted server-side. */
|
||||
export const PLATFORM_APPLICATION = "pi";
|
||||
|
||||
const PLUGIN_VERSION = (() => {
|
||||
try {
|
||||
return JSON.parse(
|
||||
fs.readFileSync(new URL("../package.json", import.meta.url), "utf-8"),
|
||||
).version as string;
|
||||
} catch {
|
||||
return "unknown";
|
||||
}
|
||||
})();
|
||||
|
||||
const MAX_STACK_ENTRIES = 4;
|
||||
const MAX_STACK_CHARS = 200;
|
||||
|
||||
/**
|
||||
* Append our own entry and bound the result, dropping WHOLE entries.
|
||||
*
|
||||
* Neither cap cuts characters: slicing the joined string severs an identifier
|
||||
* and leaves a fragment the platform parses as a real client name. And the
|
||||
* reserved slot is ours, since it is the only entry this layer can vouch for.
|
||||
*/
|
||||
function boundedStack(callerEntries: string[], own: string): string {
|
||||
const kept: string[] = [];
|
||||
let budget = MAX_STACK_CHARS - own.length;
|
||||
for (const entry of callerEntries.slice(0, MAX_STACK_ENTRIES - 1)) {
|
||||
const cost = entry.length + ", ".length;
|
||||
if (cost > budget) break;
|
||||
budget -= cost;
|
||||
kept.push(entry);
|
||||
}
|
||||
return [...kept, own].join(", ");
|
||||
}
|
||||
|
||||
/**
|
||||
* Stamp surface identity onto the shared client, once, at construction.
|
||||
*
|
||||
* Tagging individual call sites was not enough: automatic recall, capture, the
|
||||
* memory tools and deletion all go through this same client, so everything
|
||||
* except the explicit slash commands reached the platform as generic SDK
|
||||
* traffic. Every request method in the SDK sends `this.headers`, so setting
|
||||
* them here covers all of them.
|
||||
*
|
||||
* X-Mem0-Source and X-Application are set-once, so a wrapper that already named
|
||||
* a surface keeps it. X-Mem0-Client is append-only, so the platform sees the
|
||||
* whole chain rather than only the last speaker.
|
||||
*/
|
||||
export function applySurfaceHeaders(client: MemoryClient): void {
|
||||
const headers = client.headers as Record<string, string>;
|
||||
if (!headers["X-Mem0-Source"]?.trim()) headers["X-Mem0-Source"] = PLATFORM_SOURCE;
|
||||
if (!headers["X-Application"]?.trim()) headers["X-Application"] = PLATFORM_APPLICATION;
|
||||
|
||||
const existing = (headers["X-Mem0-Client"] ?? "")
|
||||
.split(",")
|
||||
.map((part) => part.trim())
|
||||
.filter(Boolean);
|
||||
headers["X-Mem0-Client"] = boundedStack(existing, `mem0-pi-agent/${PLUGIN_VERSION}`);
|
||||
}
|
||||
@@ -1,3 +1,4 @@
|
||||
import type { SearchMemoryOptions } from "mem0ai";
|
||||
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
||||
import type MemoryClient from "mem0ai";
|
||||
import type { Mem0Config, ScopeContext, Scope } from "./types.ts";
|
||||
@@ -5,6 +6,12 @@ import { DEFAULT_CUSTOM_CATEGORIES } from "./types.ts";
|
||||
import { resolveSearchFilters, resolveAddParams } from "./memory/scoping.ts";
|
||||
import { formatMemoryList, formatMemoryCompact, groupByCategory } from "./memory/formatting.ts";
|
||||
import { captureCommandEvent } from "./telemetry.ts";
|
||||
import { PLATFORM_SOURCE } from "./attribution.ts";
|
||||
|
||||
// Wire identity is set once on the shared client in entry.ts, which covers
|
||||
// every path including recall, capture, tools and deletion. It stays in the
|
||||
// body of the two calls below as well: body `source` is what the backend reads
|
||||
// when the header is absent.
|
||||
|
||||
const SEARCH_TOP_K = 10;
|
||||
|
||||
@@ -29,7 +36,12 @@ export function registerCommands(
|
||||
threshold: config.searchThreshold,
|
||||
topK: SEARCH_TOP_K,
|
||||
rerank: true,
|
||||
});
|
||||
source: PLATFORM_SOURCE,
|
||||
// Widened by exactly this one property. `source` reaches the wire via the
|
||||
// SDK's camelToSnakeKeys spread, but it is absent from SearchMemoryOptions
|
||||
// in the published mem0ai types. A blanket `as never` would also disable
|
||||
// checking of filters, threshold, topK and rerank above.
|
||||
} as SearchMemoryOptions & { source: string });
|
||||
return result.results ?? [];
|
||||
};
|
||||
|
||||
@@ -45,7 +57,7 @@ export function registerCommands(
|
||||
const addParams = resolveAddParams(config.defaultScope, getScopeCtx());
|
||||
const result = await mem0.add(
|
||||
[{ role: "user", content: text }],
|
||||
{ ...addParams, customCategories: DEFAULT_CUSTOM_CATEGORIES, infer: false },
|
||||
{ ...addParams, customCategories: DEFAULT_CUSTOM_CATEGORIES, infer: false, source: PLATFORM_SOURCE },
|
||||
);
|
||||
captureCommandEvent("mem0-remember", {}, telemetryCtx);
|
||||
|
||||
|
||||
@@ -10,6 +10,7 @@ import { captureEvent } from "./telemetry.ts";
|
||||
import * as os from "node:os";
|
||||
import type { ScopeContext } from "./types.ts";
|
||||
import { createMemoryLifecycle } from "../../agent-plugin-core/typescript/src/lifecycle.ts";
|
||||
import { applySurfaceHeaders } from "./attribution.ts";
|
||||
|
||||
export { buildRecallContext } from "../../agent-plugin-core/typescript/src/lifecycle.ts";
|
||||
|
||||
@@ -29,6 +30,11 @@ export default function mem0Extension(pi: ExtensionAPI): void {
|
||||
}
|
||||
|
||||
const mem0 = new MemoryClient({ apiKey: config.apiKey });
|
||||
// Every path below shares this client: automatic recall, capture, the memory
|
||||
// tools and deletion as well as the slash commands. Attribution belongs here
|
||||
// rather than on individual calls, or everything except the commands reports
|
||||
// as generic SDK traffic.
|
||||
applySurfaceHeaders(mem0);
|
||||
|
||||
const scopeCtx: ScopeContext = {
|
||||
userId: resolveUserId(config.userId),
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@mem0/vercel-ai-provider",
|
||||
"version": "3.0.2",
|
||||
"version": "3.0.3",
|
||||
"description": "Vercel AI Provider for providing memory to LLMs",
|
||||
"main": "./dist/index.js",
|
||||
"module": "./dist/index.mjs",
|
||||
|
||||
@@ -1,3 +1,10 @@
|
||||
declare const __MEM0_PROVIDER_VERSION__: string | undefined;
|
||||
|
||||
// Replaced at build time by tsup `define`. The fallback only applies when the
|
||||
// source is run unbundled, such as in tests.
|
||||
const PROVIDER_VERSION =
|
||||
typeof __MEM0_PROVIDER_VERSION__ !== "undefined" ? __MEM0_PROVIDER_VERSION__ : "dev";
|
||||
|
||||
import { LanguageModelV3Prompt } from '@ai-sdk/provider';
|
||||
import { Mem0ConfigSettings } from './mem0-types';
|
||||
import { loadApiKey } from '@ai-sdk/provider-utils';
|
||||
@@ -277,7 +284,11 @@ const searchInternalMemories = async (query: string, config?: Mem0ConfigSettings
|
||||
method: 'POST',
|
||||
headers: {
|
||||
Authorization: `Token ${apiKey}`,
|
||||
'Content-Type': 'application/json'
|
||||
'Content-Type': 'application/json',
|
||||
// Surface attribution. Set-once by contract: this wrapper is the
|
||||
// outermost layer on these raw fetch calls.
|
||||
'X-Mem0-Source': 'VERCEL_AI_SDK',
|
||||
'X-Mem0-Client': `mem0-vercel-ai-provider/${PROVIDER_VERSION}`
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
};
|
||||
@@ -331,7 +342,11 @@ const updateMemories = async (messages: Array<Message>, config?: Mem0ConfigSetti
|
||||
method: 'POST',
|
||||
headers: {
|
||||
Authorization: `Token ${apiKey}`,
|
||||
'Content-Type': 'application/json'
|
||||
'Content-Type': 'application/json',
|
||||
// Surface attribution. Set-once by contract: this wrapper is the
|
||||
// outermost layer on these raw fetch calls.
|
||||
'X-Mem0-Source': 'VERCEL_AI_SDK',
|
||||
'X-Mem0-Client': `mem0-vercel-ai-provider/${PROVIDER_VERSION}`
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
};
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
"noUnusedLocals": false,
|
||||
"noUnusedParameters": false,
|
||||
"preserveWatchOutput": true,
|
||||
"resolveJsonModule": true,
|
||||
"skipLibCheck": true,
|
||||
"strict": true,
|
||||
"types": ["@types/node", "jest"],
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { defineConfig } from 'tsup'
|
||||
import pkg from './package.json'
|
||||
|
||||
export default defineConfig([
|
||||
{
|
||||
@@ -6,5 +7,11 @@ export default defineConfig([
|
||||
entry: ['src/index.ts'],
|
||||
format: ['cjs', 'esm'],
|
||||
sourcemap: true,
|
||||
// Injected rather than written in the source. A hardcoded literal matches
|
||||
// package.json on the day it is written and misreports the client version
|
||||
// from the next release bump onwards. Same mechanism as mem0-ts.
|
||||
define: {
|
||||
__MEM0_PROVIDER_VERSION__: JSON.stringify(pkg.version),
|
||||
},
|
||||
},
|
||||
])
|
||||
@@ -32,7 +32,7 @@ pnpm test:unit # offline unit tests (mocked, no network)
|
||||
MEM0_API_KEY=m0-... pnpm test # unit + live E2E against api.mem0.ai
|
||||
```
|
||||
|
||||
Anonymous usage telemetry is sent to Mem0; opt out with `MEM0_TELEMETRY=false`.
|
||||
This app sends no telemetry of its own. Its API requests carry `source: "ZAPIER"` so Mem0 can see aggregate usage of the integration.
|
||||
|
||||
To deploy (maintainers): `pnpm build && zapier push`.
|
||||
|
||||
|
||||
+1
-1
@@ -13,7 +13,7 @@
|
||||
},
|
||||
"category": "Productivity",
|
||||
"description": "Cross-session memory and token savings for coding agents.",
|
||||
"version": "0.3.1"
|
||||
"version": "0.3.2"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "mem0ai",
|
||||
"version": "3.1.8",
|
||||
"version": "3.2.0",
|
||||
"description": "The Memory Layer For Your AI Apps",
|
||||
"main": "./dist/index.js",
|
||||
"module": "./dist/index.mjs",
|
||||
|
||||
@@ -95,6 +95,66 @@ interface ClientIdentity {
|
||||
const IDENTITY_CACHE_MAX_DEFAULT = 50;
|
||||
const identityByCredentials = new Map<string, Promise<ClientIdentity>>();
|
||||
|
||||
declare const __MEM0_SDK_VERSION__: string | undefined;
|
||||
|
||||
// Injected by tsup (see mem0-ts/tsup.config.ts `define`), the same mechanism
|
||||
// telemetry.ts already uses. A hardcoded literal goes stale at the next release
|
||||
// bump and then misreports the client version forever.
|
||||
const SDK_VERSION =
|
||||
typeof __MEM0_SDK_VERSION__ !== "undefined" ? __MEM0_SDK_VERSION__ : "dev";
|
||||
|
||||
const MAX_STACK_ENTRIES = 4;
|
||||
const MAX_STACK_CHARS = 200;
|
||||
|
||||
/**
|
||||
* Append our own entry and bound the result, dropping WHOLE entries.
|
||||
*
|
||||
* Neither cap cuts characters: slicing the joined string severs an identifier
|
||||
* and leaves a fragment the platform parses as a real client name. And the
|
||||
* reserved slot is ours. Pushing first and then trimming to four dropped exactly
|
||||
* the entry this exists to add whenever a caller already sent four, so we
|
||||
* vanished from our own stack while every caller claim survived.
|
||||
*/
|
||||
function boundedStack(callerEntries: string[], own: string): string {
|
||||
const kept: string[] = [];
|
||||
let budget = MAX_STACK_CHARS - own.length;
|
||||
for (const entry of callerEntries.slice(0, MAX_STACK_ENTRIES - 1)) {
|
||||
const cost = entry.length + ", ".length;
|
||||
if (cost > budget) break;
|
||||
budget -= cost;
|
||||
kept.push(entry);
|
||||
}
|
||||
return [...kept, own].join(", ");
|
||||
}
|
||||
|
||||
/**
|
||||
* Surface-identity headers.
|
||||
*
|
||||
* X-Mem0-Source and X-Application are SET-ONCE by contract: whichever layer is
|
||||
* outermost sets them and nothing below overwrites, so a plugin wrapping this
|
||||
* SDK keeps its own identity. X-Mem0-Client is APPEND-ONLY - every layer adds
|
||||
* itself, so the platform sees the whole stack and not just the last speaker.
|
||||
*/
|
||||
function surfaceHeaders(): Record<string, string> {
|
||||
const env: Record<string, string | undefined> =
|
||||
typeof process !== "undefined" && process.env ? process.env : {};
|
||||
const existing = (env.MEM0_CLIENT_STACK ?? "").trim();
|
||||
const entries = existing
|
||||
? existing
|
||||
.split(",")
|
||||
.map((part) => part.trim())
|
||||
.filter(Boolean)
|
||||
: [];
|
||||
const headers: Record<string, string> = {
|
||||
"X-Mem0-Client": boundedStack(entries, `mem0-js/${SDK_VERSION}`),
|
||||
};
|
||||
const source = (env.MEM0_SOURCE ?? "").trim();
|
||||
if (source) headers["X-Mem0-Source"] = source;
|
||||
const application = (env.MEM0_APPLICATION ?? "").trim();
|
||||
if (application) headers["X-Application"] = application;
|
||||
return headers;
|
||||
}
|
||||
|
||||
export default class MemoryClient {
|
||||
apiKey: string;
|
||||
host: string;
|
||||
@@ -129,6 +189,7 @@ export default class MemoryClient {
|
||||
this.headers = {
|
||||
Authorization: `Token ${this.apiKey}`,
|
||||
"Content-Type": "application/json",
|
||||
...surfaceHeaders(),
|
||||
};
|
||||
|
||||
this.client = axios.create({
|
||||
|
||||
@@ -30,6 +30,9 @@ export interface SearchMemoryOptions {
|
||||
showExpired?: boolean;
|
||||
referenceDate?: string | number;
|
||||
keywordSearch?: boolean;
|
||||
/** Surface that produced the call, e.g. "OPENCLAW". Must be a value the
|
||||
* backend's EventSource enum knows, or it buckets into OTHERS. */
|
||||
source?: string;
|
||||
}
|
||||
|
||||
export interface GetAllMemoryOptions {
|
||||
|
||||
+94
-24
@@ -79,6 +79,95 @@ def _maybe_alias_anon_to_email(user_email):
|
||||
logger.debug("Failed to alias anon telemetry to %r: %s", user_email, e)
|
||||
|
||||
|
||||
def _sdk_version() -> str:
|
||||
"""Resolved here rather than imported from the package root, which would cycle."""
|
||||
try:
|
||||
import importlib.metadata
|
||||
|
||||
return importlib.metadata.version("mem0ai")
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _apply_client_headers(client: Any, api_key: str, user_id: str) -> None:
|
||||
"""Merge our headers into a caller-supplied client without erasing theirs.
|
||||
|
||||
A wrapper may hand us a client already carrying its own X-Mem0-Source or a
|
||||
partial X-Mem0-Client stack. Blanket update() replaced both, which is the
|
||||
opposite of the set-once / append-only contract: the outermost layer is the
|
||||
one whose identity should survive.
|
||||
"""
|
||||
existing = client.headers
|
||||
mine = _client_headers(api_key, user_id)
|
||||
|
||||
outer_stack = existing.get("X-Mem0-Client")
|
||||
if outer_stack:
|
||||
entries = [part.strip() for part in str(outer_stack).split(",") if part.strip()]
|
||||
mine["X-Mem0-Client"] = _bounded_stack(entries, f"mem0-python/{_sdk_version()}")
|
||||
|
||||
for name, value in mine.items():
|
||||
if name in ("X-Mem0-Source", "X-Application") and existing.get(name):
|
||||
continue
|
||||
existing[name] = value
|
||||
|
||||
|
||||
MAX_STACK_ENTRIES = 4
|
||||
MAX_STACK_CHARS = 200
|
||||
|
||||
|
||||
def _bounded_stack(caller_entries, own: str) -> str:
|
||||
"""Append our own entry and bound the result, dropping WHOLE entries.
|
||||
|
||||
Two rules, and the second is the one that was wrong. Neither cap cuts
|
||||
characters: a blunt slice severs an identifier and leaves a fragment that
|
||||
parses as a real client name. And the reserved slot is OURS. Appending first
|
||||
and then trimming to four dropped exactly the entry this function exists to
|
||||
add, every time a caller already sent four, so the SDK vanished from its own
|
||||
stack while the caller's claims all survived.
|
||||
"""
|
||||
kept = []
|
||||
budget = MAX_STACK_CHARS - len(own)
|
||||
for entry in list(caller_entries)[: MAX_STACK_ENTRIES - 1]:
|
||||
cost = len(entry) + len(", ")
|
||||
if cost > budget:
|
||||
break
|
||||
budget -= cost
|
||||
kept.append(entry)
|
||||
return ", ".join(kept + [own])
|
||||
|
||||
|
||||
def _client_headers(api_key: str, user_id: str) -> Dict[str, str]:
|
||||
"""Auth plus surface-identity headers.
|
||||
|
||||
X-Mem0-Source and X-Application are SET-ONCE by contract: whichever layer is
|
||||
outermost sets them, and nothing below overwrites. A plugin or harness that
|
||||
wraps this SDK therefore keeps its own identity — it declares via MEM0_SOURCE
|
||||
/ MEM0_APPLICATION and the SDK defers.
|
||||
|
||||
X-Mem0-Client is APPEND-ONLY: every layer adds itself, so the platform sees
|
||||
the whole stack rather than only whoever spoke last.
|
||||
"""
|
||||
headers = {
|
||||
"Authorization": f"Token {api_key}",
|
||||
"Mem0-User-ID": user_id,
|
||||
"X-Mem0-Client": _client_stack(),
|
||||
}
|
||||
source = os.getenv("MEM0_SOURCE", "").strip()
|
||||
if source:
|
||||
headers["X-Mem0-Source"] = source
|
||||
application = os.getenv("MEM0_APPLICATION", "").strip()
|
||||
if application:
|
||||
headers["X-Application"] = application
|
||||
return headers
|
||||
|
||||
|
||||
def _client_stack() -> str:
|
||||
"""This SDK appended to any stack an outer layer already declared."""
|
||||
existing = os.getenv("MEM0_CLIENT_STACK", "").strip()
|
||||
entries = [part.strip() for part in existing.split(",") if part.strip()] if existing else []
|
||||
return _bounded_stack(entries, f"mem0-python/{_sdk_version()}")
|
||||
|
||||
|
||||
class MemoryClient:
|
||||
"""Client for interacting with the Mem0 API.
|
||||
|
||||
@@ -129,19 +218,11 @@ class MemoryClient:
|
||||
self.client = client
|
||||
# Ensure the client has the correct base_url and headers
|
||||
self.client.base_url = httpx.URL(self.host)
|
||||
self.client.headers.update(
|
||||
{
|
||||
"Authorization": f"Token {self.api_key}",
|
||||
"Mem0-User-ID": self.user_id,
|
||||
}
|
||||
)
|
||||
_apply_client_headers(self.client, self.api_key, self.user_id)
|
||||
else:
|
||||
self.client = httpx.Client(
|
||||
base_url=self.host,
|
||||
headers={
|
||||
"Authorization": f"Token {self.api_key}",
|
||||
"Mem0-User-ID": self.user_id,
|
||||
},
|
||||
headers=_client_headers(self.api_key, self.user_id),
|
||||
timeout=300,
|
||||
)
|
||||
self.user_email = self._validate_api_key()
|
||||
@@ -1018,19 +1099,11 @@ class AsyncMemoryClient:
|
||||
self.async_client = client
|
||||
# Ensure the client has the correct base_url and headers
|
||||
self.async_client.base_url = httpx.URL(self.host)
|
||||
self.async_client.headers.update(
|
||||
{
|
||||
"Authorization": f"Token {self.api_key}",
|
||||
"Mem0-User-ID": self.user_id,
|
||||
}
|
||||
)
|
||||
_apply_client_headers(self.async_client, self.api_key, self.user_id)
|
||||
else:
|
||||
self.async_client = httpx.AsyncClient(
|
||||
base_url=self.host,
|
||||
headers={
|
||||
"Authorization": f"Token {self.api_key}",
|
||||
"Mem0-User-ID": self.user_id,
|
||||
},
|
||||
headers=_client_headers(self.api_key, self.user_id),
|
||||
timeout=300,
|
||||
)
|
||||
|
||||
@@ -1053,10 +1126,7 @@ class AsyncMemoryClient:
|
||||
params = self._prepare_params()
|
||||
response = requests.get(
|
||||
f"{self.host}/v1/ping/",
|
||||
headers={
|
||||
"Authorization": f"Token {self.api_key}",
|
||||
"Mem0-User-ID": self.user_id,
|
||||
},
|
||||
headers=_client_headers(self.api_key, self.user_id),
|
||||
params=params,
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "mem0ai"
|
||||
version = "2.0.20"
|
||||
version = "2.1.0"
|
||||
description = "Long-term memory for AI Agents"
|
||||
authors = [
|
||||
{ name = "Mem0", email = "support@mem0.ai" }
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Surface-identity headers, and that a client can be constructed at all.
|
||||
|
||||
The construction test exists because it was not there: a signature change to
|
||||
_bounded_stack missed the _client_stack call site, every MemoryClient(...) raised
|
||||
TypeError, and the whole suite stayed green because nothing built one.
|
||||
"""
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
from mem0.client.main import _bounded_stack, _client_headers, _client_stack
|
||||
|
||||
|
||||
def test_a_client_can_be_constructed():
|
||||
from mem0 import MemoryClient
|
||||
|
||||
# _validate_api_key normally populates org/project from the API response;
|
||||
# stubbing it leaves them None, which a later accessor rejects. Set them the
|
||||
# way a real validation would. This test is about construction reaching the
|
||||
# header stage at all.
|
||||
def _stub(self):
|
||||
self.org_id, self.project_id = "org", "proj"
|
||||
|
||||
with patch.object(MemoryClient, "_validate_api_key", _stub):
|
||||
client = MemoryClient(api_key="m0-test")
|
||||
|
||||
assert client.client.headers["X-Mem0-Client"].startswith("mem0-python/")
|
||||
|
||||
|
||||
# AsyncMemoryClient is deliberately not constructed here: its validation path
|
||||
# makes a real request to /v1/ping/, and a unit test that needs the network is
|
||||
# worse than none. It shares _client_headers with the sync client, which is the
|
||||
# code the construction test above actually guards.
|
||||
def test_headers_carry_this_sdk():
|
||||
headers = _client_headers("m0-test", "u1")
|
||||
assert headers["X-Mem0-Client"].startswith("mem0-python/")
|
||||
|
||||
|
||||
def test_our_entry_survives_a_caller_that_already_filled_the_stack():
|
||||
# Appending first and trimming to four dropped exactly the entry the
|
||||
# function exists to add.
|
||||
stack = _bounded_stack(["a/1", "b/2", "c/3", "d/4"], "mem0-python/9.9.9")
|
||||
|
||||
assert "mem0-python/9.9.9" in stack
|
||||
assert len(stack.split(",")) <= 4
|
||||
|
||||
|
||||
def test_the_character_cap_drops_whole_entries_not_characters():
|
||||
long_entries = [f"{'n' * 90}/1.0", f"{'m' * 90}/1.0", "c/3"]
|
||||
stack = _bounded_stack(long_entries, "mem0-python/9.9.9")
|
||||
|
||||
assert len(stack) <= 200
|
||||
assert stack.endswith("mem0-python/9.9.9")
|
||||
for entry in stack.split(","):
|
||||
assert entry.strip().count("/") == 1, f"severed entry: {entry!r}"
|
||||
|
||||
|
||||
def test_an_outer_stack_is_appended_to_not_replaced():
|
||||
with patch.dict(os.environ, {"MEM0_CLIENT_STACK": "openclaw/2.1.0"}):
|
||||
stack = _client_stack()
|
||||
|
||||
assert stack.startswith("openclaw/2.1.0")
|
||||
assert "mem0-python/" in stack
|
||||
Reference in New Issue
Block a user