2c885fdcd7
Review found the headline fix inverted: code.install could never fire, so every fresh install reported an upgrade and the two cohorts became indistinguishable — strictly worse than the bug being fixed. hook_runner reaches claim_install() only after cache_plugin_api_key() has written `api-key` and EvidenceStore() has created `evidence.sqlite3` and its WAL files. Asking "is the data directory empty" at that point always saw content. The caller now snapshots emptiness at the top of the run, before anything writes, and passes it in. Also caught by review, all in the same file: - claim_version_change was an unsynchronized read-modify-write, so several concurrently starting sessions each observed the old version and each recorded an upgrade. The first session after a version bump is exactly when a user's open agent windows all restart together. The transition is now claimed with an exclusive per-version sentinel. - A crash between O_EXCL and the write left an empty marker, which disabled every future upgrade event on that machine: claim_install saw the file and claim_version_change could not parse it. An unparseable marker is now repaired. - claim_install consumed the one-shot claim even under MEM0_TELEMETRY=false, so a user who opted out for their first sessions would never report install after opting in. - Existing users have an email but no key fingerprint, so the fast path always missed and every flush paid an uncached /v1/ping/ — a 5s timeout each time for the offline users this stack keeps citing. Legacy rows now adopt the current key's fingerprint instead of re-resolving. - A key that will not resolve (revoked, offline) kept attributing to the previous account's email, which is the bug this was meant to fix. It now falls back to the anonymous id. - The anonymous id was never rotated, so once it had been merged into one account it was still offered as the alias for the next one. An alias naming an already-identified id is what could link two real people; it is now offered once. The gap that let this ship was that no test drove hook_runner's session-start path — the decision was only ever tested by calling claim_install() directly on a directory nothing had touched. Adds subprocess tests that run the real entrypoint: fresh install, exactly-once, and an existing data dir. 62 core tests, 203 host tests. Claude-Session: https://claude.ai/code/session_01C7tEmH86HAr7GoAAKCEHZb
764 lines
26 KiB
Python
764 lines
26 KiB
Python
#!/usr/bin/env python3
|
|
"""Usage telemetry for Mem0 agent plugins.
|
|
|
|
Events are linked to your Mem0 account email when an API key is configured, and
|
|
to a random per-machine id otherwise. Not anonymous — the Python SDK and CLI
|
|
attribute the same way.
|
|
|
|
Hooks run on a 3-6 second budget and fire on every tool call, so recording never
|
|
touches the network: `record` appends one JSON line to a local spool and returns.
|
|
A detached `python3 telemetry.py` drains the spool in one batched PostHog request,
|
|
started once per session and again from the flush worker that is already detached.
|
|
|
|
Pure stdlib, matching the rest of the plugin. Opt out with MEM0_TELEMETRY=false.
|
|
|
|
Never sends prompts, memory text, queries, file paths, repository names, or API
|
|
keys: only event names, durations, counts, coarse outcomes, and repo/session
|
|
identifiers hashed with a random per-install salt.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import platform
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
import urllib.error
|
|
import urllib.request
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import memory_core
|
|
|
|
# Seeded from the per-host module the build generates into core/. Two processes
|
|
# in this pipeline never call init() — mcp_server.py, and the detached
|
|
# `python3 telemetry.py` sender that spawn_flush() starts — so a module default
|
|
# was what every one of their events got labelled with.
|
|
try: # pragma: no cover - absent only in the un-built shared source tree
|
|
from _harness_id import HARNESS_ID as _DEFAULT_HARNESS
|
|
from _harness_id import PLATFORM_APPLICATION as _PLATFORM_APPLICATION
|
|
from _harness_id import PLATFORM_SOURCE as _PLATFORM_SOURCE
|
|
from _harness_id import SOURCE_TAG as _DEFAULT_SOURCE_TAG
|
|
except ImportError:
|
|
_DEFAULT_HARNESS = "generic"
|
|
_DEFAULT_SOURCE_TAG = "MEM0_PLUGIN"
|
|
_PLATFORM_SOURCE = "MEM0_PLUGIN"
|
|
_PLATFORM_APPLICATION = ""
|
|
|
|
_harness: str = _DEFAULT_HARNESS
|
|
_source_tag: str = _DEFAULT_SOURCE_TAG
|
|
_PRIVATE_KEYS = {
|
|
"apikey",
|
|
"authorization",
|
|
"password",
|
|
"query",
|
|
"secret",
|
|
"prompt",
|
|
"token",
|
|
"text",
|
|
"memory",
|
|
"message",
|
|
"error",
|
|
"path",
|
|
"cwd",
|
|
"userid",
|
|
"agentid",
|
|
"runid",
|
|
"repoid",
|
|
"repositoryid",
|
|
"projectid",
|
|
"appid",
|
|
"filters",
|
|
}
|
|
|
|
|
|
def init(harness: str = "", source_tag: str = "") -> None:
|
|
"""Override the generated identity. Optional — core/_harness_id.py is the default.
|
|
|
|
The fallback shape matches memory_core.configure_harness's (``<HOST>_PLUGIN``).
|
|
It used to be ``MEM0_<HOST>_PLUGIN`` here and ``<host>_plugin`` there, which
|
|
meant one plugin could emit three different source values depending on which
|
|
process happened to send the batch.
|
|
"""
|
|
global _harness, _source_tag
|
|
_harness = harness or _DEFAULT_HARNESS
|
|
_source_tag = source_tag or (
|
|
f"{_harness.upper().replace('-', '_')}_PLUGIN" if harness else _DEFAULT_SOURCE_TAG
|
|
)
|
|
|
|
POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX"
|
|
POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/"
|
|
POSTHOG_BATCH_URL = "https://us.i.posthog.com/batch/"
|
|
EVENT_PREFIX = "code"
|
|
SPOOL_LIMIT_BYTES = 256 * 1024
|
|
BATCH_SIZE = 100
|
|
SEND_TIMEOUT = 5
|
|
CLAIM_STALE_SECONDS = 120
|
|
CLAIM_EXPIRY_SECONDS = 7 * 24 * 60 * 60
|
|
# A batch is only discarded once it has genuinely been retried this many times.
|
|
MAX_CLAIM_ATTEMPTS = 3
|
|
# Parked claims drained per run, after the live spool. Bounded so a long backlog
|
|
# cannot turn one flush into an unbounded send loop.
|
|
MAX_PARKED_PER_RUN = 3
|
|
|
|
|
|
def is_enabled() -> bool:
|
|
"""Whether telemetry is switched on for this process."""
|
|
return os.environ.get("MEM0_TELEMETRY", "true").strip().lower() not in {
|
|
"false",
|
|
"0",
|
|
"no",
|
|
"off",
|
|
}
|
|
|
|
|
|
def _digest(value: str, length: int = 16) -> str:
|
|
"""Unsalted digest. Only for values that are already secrets (API keys)."""
|
|
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:length]
|
|
|
|
|
|
def _install_salt() -> str:
|
|
"""Random per-install salt, created on first use and kept in the identity file."""
|
|
identity = _read_identity()
|
|
salt = identity.get("salt")
|
|
if not salt:
|
|
salt = uuid.uuid4().hex
|
|
identity["salt"] = salt
|
|
_write_identity(identity)
|
|
return salt
|
|
|
|
|
|
def _scoped_digest(value: str, length: int = 16) -> str:
|
|
"""Salted digest for values drawn from a guessable space.
|
|
|
|
repo.identity is a git remote URL, or ``local:<absolute path>`` when there is
|
|
no remote — which normally contains the account username. Sixteen unsalted
|
|
hex characters over that input space is enumerable, so this is not a
|
|
privacy control without the salt. Salting per install keeps every
|
|
within-account join the analytics actually use and gives up only
|
|
cross-machine joins on the same repository, which nothing computes.
|
|
"""
|
|
if not value:
|
|
return ""
|
|
return hashlib.sha256(f"{_install_salt()}:{value}".encode("utf-8")).hexdigest()[:length]
|
|
|
|
|
|
def _safe_value(value: Any) -> Any:
|
|
if isinstance(value, str):
|
|
return memory_core.redact(value)
|
|
if isinstance(value, dict):
|
|
return {
|
|
key: _safe_value(item)
|
|
for key, item in value.items()
|
|
if "".join(character for character in str(key).lower() if character.isalnum())
|
|
not in _PRIVATE_KEYS
|
|
}
|
|
if isinstance(value, (list, tuple)):
|
|
return [_safe_value(item) for item in value]
|
|
if value is None or isinstance(value, (bool, int, float)):
|
|
return value
|
|
return memory_core.redact(value)
|
|
|
|
|
|
def _spool_path() -> Path:
|
|
return memory_core.data_dir() / "telemetry.jsonl"
|
|
|
|
|
|
def _identity_path() -> Path:
|
|
return memory_core.data_dir() / "telemetry-identity.json"
|
|
|
|
|
|
def _read_identity() -> dict[str, str]:
|
|
try:
|
|
value = json.loads(_identity_path().read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
return {}
|
|
return value if isinstance(value, dict) else {}
|
|
|
|
|
|
def _write_identity(identity: dict[str, str]) -> None:
|
|
path = _identity_path()
|
|
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
|
try:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
temporary.write_text(json.dumps(identity), encoding="utf-8")
|
|
temporary.replace(path)
|
|
except OSError:
|
|
try:
|
|
temporary.unlink()
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def anonymous_id(identity: dict[str, str] | None = None) -> str:
|
|
"""Per-machine anonymous identifier, created and persisted on first use."""
|
|
identity = _read_identity() if identity is None else identity
|
|
existing = identity.get("anonymous_id")
|
|
if existing:
|
|
return existing
|
|
created = f"code-anon-{uuid.uuid4().hex}"
|
|
identity["anonymous_id"] = created
|
|
_write_identity(identity)
|
|
return created
|
|
|
|
|
|
def _install_state_path() -> Path:
|
|
return memory_core.data_dir() / "install-state.json"
|
|
|
|
|
|
def is_first_run() -> bool:
|
|
"""Whether install has never been recorded on this machine.
|
|
|
|
Deliberately NOT the identity file. That file is only written by a
|
|
successful flush, so an offline or firewalled user recorded code.install on
|
|
every single session, forever — and every 0.2.x user recorded one on their
|
|
first 0.3.x session because 0.2.x never wrote it at all.
|
|
"""
|
|
return not _install_state_path().exists()
|
|
|
|
|
|
def data_dir_was_empty() -> bool:
|
|
"""Whether the data directory is untouched. Call BEFORE anything writes to it.
|
|
|
|
hook_runner reaches claim_install() only after cache_plugin_api_key() has
|
|
written `api-key` and EvidenceStore() has created `evidence.sqlite3`, so
|
|
asking at claim time always saw content and every fresh install reported an
|
|
upgrade. The caller snapshots this at the top of the run instead.
|
|
"""
|
|
return not _data_dir_has_content()
|
|
|
|
|
|
def claim_install(was_empty: bool | None = None) -> str | None:
|
|
"""Claim the one install/upgrade record for this machine, atomically.
|
|
|
|
Returns the event to record ("install" or "upgrade"), or None if another
|
|
session already claimed it. O_CREAT|O_EXCL so two sessions starting together
|
|
cannot both win.
|
|
|
|
`was_empty` must come from data_dir_was_empty() called before this process
|
|
wrote anything. Omitting it falls back to checking now, which is only
|
|
correct for a caller that has touched nothing.
|
|
"""
|
|
if not is_enabled():
|
|
# Never consume the one-shot claim while the user is opted out, or they
|
|
# would silently lose their install event if they later opt in.
|
|
return None
|
|
|
|
path = _install_state_path()
|
|
upgrading = not (data_dir_was_empty() if was_empty is None else was_empty)
|
|
try:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
handle = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
|
except FileExistsError:
|
|
return None
|
|
except OSError:
|
|
return None
|
|
try:
|
|
with os.fdopen(handle, "w", encoding="utf-8") as stream:
|
|
json.dump(
|
|
{
|
|
"plugin_version": memory_core.PLUGIN_VERSION,
|
|
"installed_at": memory_core.utc_now(),
|
|
"upgraded": upgrading,
|
|
},
|
|
stream,
|
|
)
|
|
except OSError:
|
|
pass
|
|
return "upgrade" if upgrading else "install"
|
|
|
|
|
|
def _data_dir_has_content() -> bool:
|
|
"""Whether anything predates this session in the plugin data directory."""
|
|
try:
|
|
for entry in memory_core.data_dir().iterdir():
|
|
if entry.name != "install-state.json":
|
|
return True
|
|
except OSError:
|
|
pass
|
|
return False
|
|
|
|
|
|
def _repair_install_state(path: Path) -> None:
|
|
"""Rewrite an unparseable marker so version tracking can resume."""
|
|
try:
|
|
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
|
temporary.write_text(
|
|
json.dumps({"plugin_version": memory_core.PLUGIN_VERSION, "repaired_at": memory_core.utc_now()}),
|
|
encoding="utf-8",
|
|
)
|
|
temporary.replace(path)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def claim_version_change() -> str | None:
|
|
"""Return the previously recorded version if it differs, updating the marker.
|
|
|
|
Only meaningful once the marker exists — the first transition into 0.3.x has
|
|
no recorded predecessor and reports "pre-0.3" instead. Claiming by rewriting
|
|
the marker means the next session sees no change and records nothing.
|
|
"""
|
|
path = _install_state_path()
|
|
try:
|
|
state = json.loads(path.read_text(encoding="utf-8"))
|
|
except OSError:
|
|
return None
|
|
except json.JSONDecodeError:
|
|
# A crash between O_EXCL and the write leaves an empty marker. Left
|
|
# alone it disables every future upgrade event on this machine, because
|
|
# claim_install sees the file and this function cannot parse it.
|
|
state = None
|
|
if not isinstance(state, dict):
|
|
_repair_install_state(path)
|
|
return None
|
|
previous = str(state.get("plugin_version") or "")
|
|
if not previous or previous == memory_core.PLUGIN_VERSION:
|
|
return None
|
|
# Claim the transition with an exclusive sentinel before rewriting the
|
|
# marker. A plain read-modify-write let every concurrently starting session
|
|
# observe the old version and each record its own upgrade — and the first
|
|
# session after a version bump is exactly when several agent windows restart
|
|
# together.
|
|
sentinel = path.with_name(f"upgraded-{memory_core.PLUGIN_VERSION}")
|
|
try:
|
|
os.close(os.open(sentinel, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600))
|
|
except FileExistsError:
|
|
return None
|
|
except OSError:
|
|
return None
|
|
|
|
state["plugin_version"] = memory_core.PLUGIN_VERSION
|
|
state["upgraded_at"] = memory_core.utc_now()
|
|
try:
|
|
temporary = path.with_suffix(f".{os.getpid()}.tmp")
|
|
temporary.write_text(json.dumps(state), encoding="utf-8")
|
|
temporary.replace(path)
|
|
except OSError:
|
|
return None
|
|
return previous
|
|
|
|
|
|
def record(
|
|
event: str,
|
|
*,
|
|
repo: Any = None,
|
|
session_id: str | None = None,
|
|
**properties: Any,
|
|
) -> None:
|
|
"""Append one event to the local spool. Never blocks and never raises."""
|
|
if not is_enabled():
|
|
return
|
|
try:
|
|
spool = _spool_path()
|
|
try:
|
|
if spool.stat().st_size > SPOOL_LIMIT_BYTES:
|
|
return
|
|
except OSError:
|
|
pass
|
|
properties = _safe_value(properties)
|
|
# Stamped in the RECORDING process, beside harness. `source` used to be
|
|
# read in the sending process from a module global, so whichever process
|
|
# drained the spool named every event in it. flush() spreads per-event
|
|
# properties last, so this now wins over any sender's default.
|
|
properties.update(
|
|
harness=_harness,
|
|
source=_source_tag,
|
|
plugin_version=memory_core.PLUGIN_VERSION,
|
|
os=sys.platform,
|
|
python_version=platform.python_version(),
|
|
)
|
|
if repo is not None:
|
|
properties["repo_hash"] = _scoped_digest(getattr(repo, "identity", ""))
|
|
if session_id:
|
|
properties["session_hash"] = _scoped_digest(session_id)
|
|
line = json.dumps(
|
|
{
|
|
"event": f"{EVENT_PREFIX}.{event}",
|
|
"uuid": str(uuid.uuid4()),
|
|
"timestamp": memory_core.utc_now(),
|
|
"properties": {
|
|
key: value for key, value in properties.items() if value is not None
|
|
},
|
|
},
|
|
separators=(",", ":"),
|
|
default=str,
|
|
)
|
|
spool.parent.mkdir(parents=True, exist_ok=True)
|
|
with spool.open("a", encoding="utf-8") as handle:
|
|
handle.write(line + "\n")
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def error_kind(exc: BaseException | str) -> str:
|
|
"""Coarse, content-free label for a failure, safe to send."""
|
|
text = exc if isinstance(exc, str) else f"{type(exc).__name__}: {exc}"
|
|
lowered = text.lower()
|
|
if "timed out" in lowered or "timeout" in lowered:
|
|
return "timeout"
|
|
if "401" in lowered or "403" in lowered or "unauthor" in lowered or "forbidden" in lowered:
|
|
return "auth"
|
|
if "429" in lowered or "rate limit" in lowered:
|
|
return "rate-limited"
|
|
if any(code in lowered for code in ("500", "502", "503", "504")):
|
|
return "server-error"
|
|
if "400" in lowered or "422" in lowered:
|
|
return "bad-request"
|
|
if isinstance(exc, str):
|
|
return "other"
|
|
if isinstance(exc, urllib.error.URLError):
|
|
return "network"
|
|
return type(exc).__name__
|
|
|
|
|
|
def spawn_flush() -> bool:
|
|
"""Start the detached sender that drains the spool."""
|
|
if not is_enabled():
|
|
return False
|
|
try:
|
|
if not _spool_path().exists() and not any(
|
|
memory_core.data_dir().glob("telemetry-*.sending")
|
|
):
|
|
return False
|
|
subprocess.Popen(
|
|
[sys.executable, str(Path(__file__).resolve())],
|
|
stdin=subprocess.DEVNULL,
|
|
stdout=subprocess.DEVNULL,
|
|
stderr=subprocess.DEVNULL,
|
|
close_fds=True,
|
|
**memory_core.detached_process_kwargs(),
|
|
)
|
|
return True
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _claim_name(attempt: int = 0) -> str:
|
|
"""Claim filename. The attempt count rides in the name so the 7-day expiry
|
|
only ever discards a batch that was actually retried and failed."""
|
|
return f"telemetry-{os.getpid()}-{uuid.uuid4().hex[:8]}-a{attempt}.sending"
|
|
|
|
|
|
def _claim_attempt(claim: Path) -> int:
|
|
"""Attempts recorded in a claim filename; 0 for the pre-attempt-count shape."""
|
|
stem = claim.name[: -len(".sending")] if claim.name.endswith(".sending") else claim.name
|
|
tail = stem.rsplit("-", 1)[-1]
|
|
if tail.startswith("a") and tail[1:].isdigit():
|
|
return int(tail[1:])
|
|
return 0
|
|
|
|
|
|
def _touch(path: Path) -> None:
|
|
"""Refresh mtime so a claim's age measures time since it was claimed.
|
|
|
|
``Path.replace`` is ``os.rename``, which preserves mtime — so a claim created
|
|
after a quiet minute inherited the spool's last-write time and looked
|
|
abandoned the instant it was made. A second sender would then take it over
|
|
while the first was still posting, and both would deliver the batch.
|
|
"""
|
|
try:
|
|
os.utime(path, None)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _claim_spool() -> Path | None:
|
|
"""Rename the spool aside so exactly one sender owns each batch."""
|
|
directory = memory_core.data_dir()
|
|
claim = directory / _claim_name()
|
|
spool = _spool_path()
|
|
try:
|
|
spool.replace(claim)
|
|
_touch(claim)
|
|
return claim
|
|
except OSError:
|
|
pass
|
|
return _claim_parked(directory)
|
|
|
|
|
|
def _claim_parked(directory: Path) -> Path | None:
|
|
"""Take the oldest abandoned claim, if any lease has actually expired.
|
|
|
|
Kept separate from the live spool so flush() can drain both in one run.
|
|
Previously parked batches were only reachable when no spool existed at all,
|
|
and because sessions keep recording there usually was one — so a batch
|
|
parked by a failed send waited until the 7-day expiry deleted it unsent,
|
|
even though its own presence is what started the sender.
|
|
"""
|
|
now = time.time()
|
|
for orphan in sorted(directory.glob("telemetry-*.sending"), key=_safe_mtime):
|
|
try:
|
|
age = now - orphan.stat().st_mtime
|
|
except OSError:
|
|
continue
|
|
if age > CLAIM_EXPIRY_SECONDS and _claim_attempt(orphan) >= MAX_CLAIM_ATTEMPTS:
|
|
try:
|
|
orphan.unlink()
|
|
except OSError:
|
|
pass
|
|
continue
|
|
if age < CLAIM_STALE_SECONDS:
|
|
# Someone else holds a live lease on it.
|
|
continue
|
|
claim = orphan.parent / _claim_name(_claim_attempt(orphan) + 1)
|
|
try:
|
|
orphan.replace(claim)
|
|
_touch(claim)
|
|
return claim
|
|
except OSError:
|
|
continue
|
|
return None
|
|
|
|
|
|
def _safe_mtime(path: Path) -> float:
|
|
try:
|
|
return path.stat().st_mtime
|
|
except OSError:
|
|
return 0.0
|
|
|
|
|
|
def _rewrite_claim(claim: Path, remaining: list[dict[str, Any]]) -> bool:
|
|
"""Persist the unsent remainder, atomically, and refresh the lease.
|
|
|
|
Called after every successful batch. Two jobs: a retry resumes where the
|
|
send stopped instead of re-posting from the top, and the rewrite doubles as
|
|
the lease heartbeat, so a slow sender does not have its claim stolen
|
|
mid-flight. Interval is one batch, well inside CLAIM_STALE_SECONDS.
|
|
"""
|
|
if not remaining:
|
|
try:
|
|
claim.unlink()
|
|
except OSError:
|
|
pass
|
|
return True
|
|
temporary = claim.with_suffix(f".{os.getpid()}.partial")
|
|
try:
|
|
temporary.write_text(
|
|
"".join(json.dumps(event, separators=(",", ":"), default=str) + "\n" for event in remaining),
|
|
encoding="utf-8",
|
|
)
|
|
temporary.replace(claim)
|
|
_touch(claim)
|
|
return True
|
|
except OSError:
|
|
try:
|
|
temporary.unlink()
|
|
except OSError:
|
|
pass
|
|
return False
|
|
|
|
|
|
def _release_claim(claim: Path, remaining: list[dict[str, Any]]) -> None:
|
|
"""Persist the remainder and drop the lease, because this sender has given up.
|
|
|
|
Distinct from the per-batch heartbeat: heartbeating on the way out would
|
|
make an abandoned batch look actively owned for a further
|
|
CLAIM_STALE_SECONDS, delaying the retry for no reason. Ageing it past the
|
|
threshold lets the next flush pick it up immediately, while the attempt
|
|
count in the filename still bounds how many times that can happen.
|
|
"""
|
|
if not _rewrite_claim(claim, remaining):
|
|
return
|
|
try:
|
|
released = time.time() - CLAIM_STALE_SECONDS - 1
|
|
os.utime(claim, (released, released))
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _resolve_email(key: str) -> str:
|
|
"""Trade the API key for the account email so events join other Mem0 surfaces."""
|
|
url = os.environ.get("MEM0_API_URL", memory_core.DEFAULT_API_URL).rstrip("/") + "/v1/ping/"
|
|
request = urllib.request.Request(
|
|
url, headers={"Authorization": f"Token {key}", "Content-Type": "application/json"}
|
|
)
|
|
try:
|
|
with urllib.request.urlopen(request, timeout=SEND_TIMEOUT) as response:
|
|
payload = json.loads(response.read().decode("utf-8"))
|
|
except Exception:
|
|
return ""
|
|
email = payload.get("user_email") if isinstance(payload, dict) else ""
|
|
return email if isinstance(email, str) else ""
|
|
|
|
|
|
def _post(payload: dict[str, Any], url: str) -> bool:
|
|
request = urllib.request.Request(
|
|
url,
|
|
data=json.dumps(payload, default=str).encode("utf-8"),
|
|
headers={"Content-Type": "application/json"},
|
|
)
|
|
try:
|
|
with urllib.request.urlopen(request, timeout=SEND_TIMEOUT):
|
|
return True
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def resolve_distinct_id() -> tuple[str, str]:
|
|
"""Return the PostHog distinct id and the anonymous id it replaced, if any.
|
|
|
|
The second value becomes a PostHog $identify alias. It is ONLY ever an
|
|
anonymous id: aliasing one account email to another merges two real person
|
|
profiles and cannot be undone, so a key that now belongs to a different
|
|
account re-resolves with no alias.
|
|
"""
|
|
identity = _read_identity()
|
|
key = memory_core.api_key()
|
|
fingerprint = _digest(key) if key else ""
|
|
email = identity.get("email", "")
|
|
|
|
if email and fingerprint:
|
|
recorded = identity.get("key_fingerprint", "")
|
|
if recorded == fingerprint:
|
|
return email, ""
|
|
if not recorded:
|
|
# Rows written before fingerprints existed. Adopt the current key
|
|
# rather than re-resolving: otherwise every existing user pays an
|
|
# uncached /v1/ping/ on every flush, forever, and a firewalled one
|
|
# pays the full timeout each time.
|
|
identity["key_fingerprint"] = fingerprint
|
|
_write_identity(identity)
|
|
return email, ""
|
|
|
|
if not key:
|
|
# No key to verify the account with; do not keep attributing to it.
|
|
if email:
|
|
identity.pop("email", None)
|
|
identity.pop("key_fingerprint", None)
|
|
_write_identity(identity)
|
|
return anonymous_id(identity), ""
|
|
|
|
resolved = _resolve_email(key)
|
|
if not resolved:
|
|
# The key changed and will not resolve (revoked, offline, API down).
|
|
# Do not keep attributing to the previous account.
|
|
return anonymous_id(identity), ""
|
|
|
|
# Alias only when going anonymous -> email for the first time. Once an anon
|
|
# id has been merged into an account it must never be offered again: an
|
|
# alias naming an already-identified id is what could link two real people.
|
|
previous = "" if (email or identity.get("aliased")) else identity.get("anonymous_id", "")
|
|
if previous:
|
|
identity["aliased"] = True
|
|
identity["email"] = resolved
|
|
identity["key_fingerprint"] = fingerprint
|
|
_write_identity(identity)
|
|
return resolved, previous
|
|
|
|
|
|
def flush() -> int:
|
|
"""Drain the live spool, then any parked claims, and return events sent."""
|
|
if not is_enabled():
|
|
return 0
|
|
sent, delivered = _drain(_claim_spool())
|
|
if not delivered:
|
|
# The network is failing. Retrying other batches now would only burn
|
|
# their attempt budget against the same broken connection.
|
|
return sent
|
|
|
|
# Parked batches used to starve behind the live spool indefinitely. Bounded
|
|
# per run so a long backlog cannot turn one flush into an unbounded loop.
|
|
directory = memory_core.data_dir()
|
|
for _ in range(MAX_PARKED_PER_RUN):
|
|
parked = _claim_parked(directory)
|
|
if parked is None:
|
|
break
|
|
count, delivered = _drain(parked)
|
|
sent += count
|
|
if not delivered:
|
|
break
|
|
return sent
|
|
|
|
|
|
def _drain(claim: Path | None) -> tuple[int, bool]:
|
|
"""Post one claimed batch file, recording progress after every batch.
|
|
|
|
Returns (events sent, whether everything was delivered).
|
|
"""
|
|
if claim is None:
|
|
return 0, True
|
|
try:
|
|
lines = claim.read_text(encoding="utf-8").splitlines()
|
|
except OSError:
|
|
return 0, True
|
|
events = []
|
|
for line in lines:
|
|
try:
|
|
value = json.loads(line)
|
|
except json.JSONDecodeError:
|
|
continue
|
|
if isinstance(value, dict) and value.get("event"):
|
|
events.append(value)
|
|
if not events:
|
|
try:
|
|
claim.unlink()
|
|
except OSError:
|
|
pass
|
|
return 0, True
|
|
|
|
distinct_id, aliased_anonymous_id = resolve_distinct_id()
|
|
if aliased_anonymous_id:
|
|
_post(
|
|
{
|
|
"api_key": POSTHOG_API_KEY,
|
|
"event": "$identify",
|
|
"distinct_id": distinct_id,
|
|
"properties": {
|
|
"$anon_distinct_id": aliased_anonymous_id,
|
|
"$lib": "posthog-python",
|
|
},
|
|
},
|
|
POSTHOG_CAPTURE_URL,
|
|
)
|
|
|
|
sent = 0
|
|
for start in range(0, len(events), BATCH_SIZE):
|
|
chunk = events[start : start + BATCH_SIZE]
|
|
batch = [
|
|
{
|
|
"event": event["event"],
|
|
"distinct_id": distinct_id,
|
|
# Carried through from record() so a resend can be collapsed.
|
|
"uuid": event.get("uuid"),
|
|
"timestamp": event.get("timestamp"),
|
|
"properties": {
|
|
# Fallback only: events recorded by a build before source
|
|
# moved into record() have none of their own.
|
|
"source": _source_tag,
|
|
"language": "python",
|
|
"$process_person_profile": False,
|
|
"$lib": "posthog-python",
|
|
**(event.get("properties") or {}),
|
|
},
|
|
}
|
|
for event in chunk
|
|
]
|
|
if not _post({"api_key": POSTHOG_API_KEY, "batch": batch}, POSTHOG_BATCH_URL):
|
|
# Keep only what has not been delivered, and release the lease.
|
|
# Previously the whole file was kept and the retry re-posted every
|
|
# batch, including the ones that had already arrived.
|
|
_release_claim(claim, events[start:])
|
|
return sent, False
|
|
sent += len(chunk)
|
|
# Record progress and refresh the lease after each successful batch, so
|
|
# a crash repeats at most one batch instead of the entire file.
|
|
_rewrite_claim(claim, events[start + len(chunk) :])
|
|
return sent, True
|
|
|
|
|
|
def main() -> int:
|
|
flush()
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
try:
|
|
raise SystemExit(main())
|
|
except Exception:
|
|
raise SystemExit(0)
|