diff --git a/docs/components/rerankers/config.mdx b/docs/components/rerankers/config.mdx new file mode 100644 index 000000000..b4bac0310 --- /dev/null +++ b/docs/components/rerankers/config.mdx @@ -0,0 +1,90 @@ +--- +title: Config +description: 'Configuration options for rerankers in Mem0' +icon: "gear" +iconType: "solid" +--- + +## Common Configuration Parameters + +All rerankers share these common configuration parameters: + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `provider` | Reranker provider name | `str` | Required | +| `top_k` | Maximum number of results to return after reranking | `int` | `None` | +| `api_key` | API key for the reranker service | `str` | `None` | + +## Provider-Specific Configuration + +### Zero Entropy + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | Model to use: `zerank-1` or `zerank-1-small` | `str` | `"zerank-1"` | +| `api_key` | Zero Entropy API key | `str` | `None` | + +### Cohere + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | Cohere rerank model | `str` | `"rerank-english-v3.0"` | +| `api_key` | Cohere API key | `str` | `None` | +| `return_documents` | Whether to return document texts in response | `bool` | `False` | +| `max_chunks_per_doc` | Maximum chunks per document | `int` | `None` | + +### Sentence Transformer + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | HuggingFace cross-encoder model name | `str` | `"cross-encoder/ms-marco-MiniLM-L-6-v2"` | +| `device` | Device to run model on (`cpu`, `cuda`, etc.) | `str` | `None` | +| `batch_size` | Batch size for processing | `int` | `32` | +| `show_progress_bar` | Show progress during processing | `bool` | `False` | + +### LLM-based + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | LLM model to use for scoring | `str` | `"gpt-4o-mini"` | +| `provider` | LLM provider (`openai`, `anthropic`, etc.) | `str` | `"openai"` | +| `api_key` | API key for LLM provider | `str` | `None` | +| `temperature` | Temperature for LLM generation | `float` | `0.0` | +| `max_tokens` | Maximum tokens for LLM response | `int` | `100` | +| `scoring_prompt` | Custom prompt template for scoring | `str` | Default scoring prompt | + +## Environment Variables + +You can set API keys using environment variables: + +- `ZERO_ENTROPY_API_KEY` - Zero Entropy API key +- `COHERE_API_KEY` - Cohere API key +- `OPENAI_API_KEY` - OpenAI API key (for LLM-based reranker) +- `ANTHROPIC_API_KEY` - Anthropic API key (for LLM-based reranker) + +## Basic Configuration Example + +```python Python +config = { + "vector_store": { + "provider": "chroma", + "config": { + "collection_name": "my_memories", + "path": "./chroma_db" + } + }, + "llm": { + "provider": "openai", + "config": { + "model": "gpt-4o-mini" + } + }, + "rerank": { + "provider": "zero_entropy", + "config": { + "model": "zerank-1", + "top_k": 5 + } + } +} +``` \ No newline at end of file diff --git a/docs/components/rerankers/models/cohere.mdx b/docs/components/rerankers/models/cohere.mdx new file mode 100644 index 000000000..db173c288 --- /dev/null +++ b/docs/components/rerankers/models/cohere.mdx @@ -0,0 +1,147 @@ +--- +title: Cohere +description: 'Enterprise-grade reranking with Cohere' +icon: "building" +iconType: "solid" +--- + +Cohere provides enterprise-grade reranking models with excellent multilingual support and production-ready performance. + +## Models + +Cohere offers several reranking models: + +- **`rerank-english-v3.0`**: Latest English reranker with best performance +- **`rerank-multilingual-v3.0`**: Multilingual support for global applications +- **`rerank-english-v2.0`**: Previous generation English reranker + +## Installation + +```bash +pip install cohere +``` + +## Configuration + +```python Python +from mem0 import Memory + +config = { + "vector_store": { + "provider": "chroma", + "config": { + "collection_name": "my_memories", + "path": "./chroma_db" + } + }, + "llm": { + "provider": "openai", + "config": { + "model": "gpt-4o-mini" + } + }, + "rerank": { + "provider": "cohere", + "config": { + "model": "rerank-english-v3.0", + "api_key": "your-cohere-api-key", # or set COHERE_API_KEY + "top_k": 5, + "return_documents": False, + "max_chunks_per_doc": None + } + } +} + +memory = Memory.from_config(config) +``` + +## Environment Variables + +Set your API key as an environment variable: + +```bash +export COHERE_API_KEY="your-api-key" +``` + +## Usage Example + +```python Python +import os +from mem0 import Memory + +# Set API key +os.environ["COHERE_API_KEY"] = "your-api-key" + +# Initialize memory with Cohere reranker +config = { + "vector_store": {"provider": "chroma"}, + "llm": {"provider": "openai", "config": {"model": "gpt-4o-mini"}}, + "rerank": { + "provider": "cohere", + "config": { + "model": "rerank-english-v3.0", + "top_k": 3 + } + } +} + +memory = Memory.from_config(config) + +# Add memories +messages = [ + {"role": "user", "content": "I work as a data scientist at Microsoft"}, + {"role": "user", "content": "I specialize in machine learning and NLP"}, + {"role": "user", "content": "I enjoy playing tennis on weekends"} +] + +memory.add(messages, user_id="bob") + +# Search with reranking +results = memory.search("What is the user's profession?", user_id="bob") + +for result in results['results']: + print(f"Memory: {result['memory']}") + print(f"Vector Score: {result['score']:.3f}") + print(f"Rerank Score: {result['rerank_score']:.3f}") + print() +``` + +## Multilingual Support + +For multilingual applications, use the multilingual model: + +```python Python +config = { + "rerank": { + "provider": "cohere", + "config": { + "model": "rerank-multilingual-v3.0", + "top_k": 5 + } + } +} +``` + +## Configuration Parameters + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | Cohere rerank model to use | `str` | `"rerank-english-v3.0"` | +| `api_key` | Cohere API key | `str` | `None` | +| `top_k` | Maximum documents to return | `int` | `None` | +| `return_documents` | Whether to return document texts | `bool` | `False` | +| `max_chunks_per_doc` | Maximum chunks per document | `int` | `None` | + +## Features + +- **High Quality**: Enterprise-grade relevance scoring +- **Multilingual**: Support for 100+ languages +- **Scalable**: Production-ready with high throughput +- **Reliable**: SLA-backed service with 99.9% uptime + +## Best Practices + +1. **Model Selection**: Use `rerank-english-v3.0` for English, `rerank-multilingual-v3.0` for other languages +2. **Batch Processing**: Process multiple queries efficiently +3. **Error Handling**: Implement retry logic for production systems +4. **Monitoring**: Track reranking performance and costs \ No newline at end of file diff --git a/docs/components/rerankers/models/llm.mdx b/docs/components/rerankers/models/llm.mdx new file mode 100644 index 000000000..faa493133 --- /dev/null +++ b/docs/components/rerankers/models/llm.mdx @@ -0,0 +1,214 @@ +--- +title: LLM-based +description: 'Flexible reranking using any Large Language Model' +icon: "robot" +iconType: "solid" +--- + +LLM-based reranker provides maximum flexibility by using any Large Language Model to score document relevance. This approach allows for custom prompts and domain-specific scoring logic. + +## Supported LLM Providers + +Any LLM provider supported by Mem0 can be used for reranking: + +- **OpenAI**: GPT-4, GPT-3.5-turbo, etc. +- **Anthropic**: Claude models +- **Together**: Open-source models +- **Groq**: Fast inference +- **Ollama**: Local models +- And more... + +## Configuration + +```python Python +from mem0 import Memory + +config = { + "vector_store": { + "provider": "chroma", + "config": { + "collection_name": "my_memories", + "path": "./chroma_db" + } + }, + "llm": { + "provider": "openai", + "config": { + "model": "gpt-4o-mini" + } + }, + "rerank": { + "provider": "llm", + "config": { + "model": "gpt-4o-mini", + "provider": "openai", + "api_key": "your-openai-api-key", # or set OPENAI_API_KEY + "top_k": 5, + "temperature": 0.0 + } + } +} + +memory = Memory.from_config(config) +``` + +## Custom Scoring Prompt + +You can provide a custom prompt for relevance scoring: + +```python Python +custom_prompt = """You are a relevance scoring assistant. Rate how well this document answers the query. + +Query: "{query}" +Document: "{document}" + +Score from 0.0 to 1.0 where: +- 1.0: Perfect match, directly answers the query +- 0.8-0.9: Highly relevant, good match +- 0.6-0.7: Moderately relevant, partial match +- 0.4-0.5: Slightly relevant, limited useful information +- 0.0-0.3: Not relevant or no useful information + +Provide only a single numerical score between 0.0 and 1.0.""" + +config["rerank"]["config"]["scoring_prompt"] = custom_prompt +``` + +## Usage Example + +```python Python +import os +from mem0 import Memory + +# Set API key +os.environ["OPENAI_API_KEY"] = "your-api-key" + +# Initialize memory with LLM reranker +config = { + "vector_store": {"provider": "chroma"}, + "llm": {"provider": "openai", "config": {"model": "gpt-4o-mini"}}, + "rerank": { + "provider": "llm", + "config": { + "model": "gpt-4o-mini", + "provider": "openai", + "temperature": 0.0 + } + } +} + +memory = Memory.from_config(config) + +# Add memories +messages = [ + {"role": "user", "content": "I'm learning Python programming"}, + {"role": "user", "content": "I find object-oriented programming challenging"}, + {"role": "user", "content": "I love hiking in national parks"} +] + +memory.add(messages, user_id="david") + +# Search with LLM reranking +results = memory.search("What programming topics is the user studying?", user_id="david") + +for result in results['results']: + print(f"Memory: {result['memory']}") + print(f"Vector Score: {result['score']:.3f}") + print(f"Rerank Score: {result['rerank_score']:.3f}") + print() +``` + +## Domain-Specific Scoring + +Create specialized scoring for your domain: + +```python Python +medical_prompt = """You are a medical relevance expert. Score how relevant this medical record is to the clinical query. + +Clinical Query: "{query}" +Medical Record: "{document}" + +Consider: +- Clinical relevance and accuracy +- Patient safety implications +- Diagnostic value +- Treatment relevance + +Score from 0.0 to 1.0. Provide only the numerical score.""" + +config = { + "rerank": { + "provider": "llm", + "config": { + "model": "gpt-4o-mini", + "provider": "openai", + "scoring_prompt": medical_prompt, + "temperature": 0.0 + } + } +} +``` + +## Multiple LLM Providers + +Use different LLM providers for reranking: + +```python Python +# Using Anthropic Claude +anthropic_config = { + "rerank": { + "provider": "llm", + "config": { + "model": "claude-3-haiku-20240307", + "provider": "anthropic", + "temperature": 0.0 + } + } +} + +# Using local Ollama model +ollama_config = { + "rerank": { + "provider": "llm", + "config": { + "model": "llama2:7b", + "provider": "ollama", + "temperature": 0.0 + } + } +} +``` + +## Configuration Parameters + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | LLM model to use for scoring | `str` | `"gpt-4o-mini"` | +| `provider` | LLM provider name | `str` | `"openai"` | +| `api_key` | API key for the LLM provider | `str` | `None` | +| `top_k` | Maximum documents to return | `int` | `None` | +| `temperature` | Temperature for LLM generation | `float` | `0.0` | +| `max_tokens` | Maximum tokens for LLM response | `int` | `100` | +| `scoring_prompt` | Custom prompt template | `str` | Default prompt | + +## Advantages + +- **Maximum Flexibility**: Custom prompts for any use case +- **Domain Expertise**: Leverage LLM knowledge for specialized domains +- **Interpretability**: Understand scoring through prompt engineering +- **Multi-criteria**: Score based on multiple relevance factors + +## Considerations + +- **Latency**: Higher latency than specialized rerankers +- **Cost**: LLM API costs per reranking operation +- **Consistency**: May have slight variations in scoring +- **Prompt Engineering**: Requires careful prompt design + +## Best Practices + +1. **Temperature**: Use 0.0 for consistent scoring +2. **Prompt Design**: Be specific about scoring criteria +3. **Token Efficiency**: Keep prompts concise to reduce costs +4. **Caching**: Cache results for repeated queries when possible +5. **Fallback**: Handle API errors gracefully \ No newline at end of file diff --git a/docs/components/rerankers/models/sentence_transformer.mdx b/docs/components/rerankers/models/sentence_transformer.mdx new file mode 100644 index 000000000..008777b43 --- /dev/null +++ b/docs/components/rerankers/models/sentence_transformer.mdx @@ -0,0 +1,161 @@ +--- +title: Sentence Transformer +description: 'Local reranking with HuggingFace cross-encoder models' +icon: "server" +iconType: "solid" +--- + +Sentence Transformer reranker provides local reranking using HuggingFace cross-encoder models, perfect for privacy-focused deployments where you want to keep data on-premises. + +## Models + +Any HuggingFace cross-encoder model can be used. Popular choices include: + +- **`cross-encoder/ms-marco-MiniLM-L-6-v2`**: Default, good balance of speed and accuracy +- **`cross-encoder/ms-marco-TinyBERT-L-2-v2`**: Fastest, smaller model size +- **`cross-encoder/ms-marco-electra-base`**: Higher accuracy, larger model +- **`cross-encoder/stsb-distilroberta-base`**: Good for semantic similarity tasks + +## Installation + +```bash +pip install sentence-transformers +``` + +## Configuration + +```python Python +from mem0 import Memory + +config = { + "vector_store": { + "provider": "chroma", + "config": { + "collection_name": "my_memories", + "path": "./chroma_db" + } + }, + "llm": { + "provider": "openai", + "config": { + "model": "gpt-4o-mini" + } + }, + "rerank": { + "provider": "sentence_transformer", + "config": { + "model": "cross-encoder/ms-marco-MiniLM-L-6-v2", + "device": "cpu", # or "cuda" for GPU + "batch_size": 32, + "show_progress_bar": False, + "top_k": 5 + } + } +} + +memory = Memory.from_config(config) +``` + +## GPU Acceleration + +For better performance, use GPU acceleration: + +```python Python +config = { + "rerank": { + "provider": "sentence_transformer", + "config": { + "model": "cross-encoder/ms-marco-MiniLM-L-6-v2", + "device": "cuda", # Use GPU + "batch_size": 64 # Larger batch size for GPU + } + } +} +``` + +## Usage Example + +```python Python +from mem0 import Memory + +# Initialize memory with local reranker +config = { + "vector_store": {"provider": "chroma"}, + "llm": {"provider": "openai", "config": {"model": "gpt-4o-mini"}}, + "rerank": { + "provider": "sentence_transformer", + "config": { + "model": "cross-encoder/ms-marco-MiniLM-L-6-v2", + "device": "cpu" + } + } +} + +memory = Memory.from_config(config) + +# Add memories +messages = [ + {"role": "user", "content": "I love reading science fiction novels"}, + {"role": "user", "content": "My favorite author is Isaac Asimov"}, + {"role": "user", "content": "I also enjoy watching sci-fi movies"} +] + +memory.add(messages, user_id="charlie") + +# Search with local reranking +results = memory.search("What books does the user like?", user_id="charlie") + +for result in results['results']: + print(f"Memory: {result['memory']}") + print(f"Vector Score: {result['score']:.3f}") + print(f"Rerank Score: {result['rerank_score']:.3f}") + print() +``` + +## Custom Models + +You can use any HuggingFace cross-encoder model: + +```python Python +# Using a different model +config = { + "rerank": { + "provider": "sentence_transformer", + "config": { + "model": "cross-encoder/stsb-distilroberta-base", + "device": "cpu" + } + } +} +``` + +## Configuration Parameters + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | HuggingFace cross-encoder model name | `str` | `"cross-encoder/ms-marco-MiniLM-L-6-v2"` | +| `device` | Device to run model on (`cpu`, `cuda`, etc.) | `str` | `None` | +| `batch_size` | Batch size for processing documents | `int` | `32` | +| `show_progress_bar` | Show progress bar during processing | `bool` | `False` | +| `top_k` | Maximum documents to return | `int` | `None` | + +## Advantages + +- **Privacy**: Complete local processing, no external API calls +- **Cost**: No per-token charges after initial model download +- **Customization**: Use any HuggingFace cross-encoder model +- **Offline**: Works without internet connection after model download + +## Performance Considerations + +- **First Run**: Model download may take time initially +- **Memory Usage**: Models require GPU/CPU memory +- **Batch Size**: Optimize batch size based on available memory +- **Device**: GPU acceleration significantly improves speed + +## Best Practices + +1. **Model Selection**: Choose model based on accuracy vs speed requirements +2. **Device Management**: Use GPU when available for better performance +3. **Batch Processing**: Process multiple documents together for efficiency +4. **Memory Monitoring**: Monitor system memory usage with larger models \ No newline at end of file diff --git a/docs/components/rerankers/models/zero_entropy.mdx b/docs/components/rerankers/models/zero_entropy.mdx new file mode 100644 index 000000000..37698b38c --- /dev/null +++ b/docs/components/rerankers/models/zero_entropy.mdx @@ -0,0 +1,119 @@ +--- +title: Zero Entropy +description: 'State-of-the-art neural reranking with Zero Entropy' +icon: "sparkles" +iconType: "solid" +--- + +[Zero Entropy](https://www.zeroentropy.dev) provides state-of-the-art neural reranking models that significantly improve search relevance with fast performance. + +## Models + +Zero Entropy offers two reranking models: + +- **`zerank-1`**: Flagship state-of-the-art reranker (non-commercial license) +- **`zerank-1-small`**: Open-source model (Apache 2.0 license) + +## Installation + +```bash +pip install zeroentropy +``` + +## Configuration + +```python Python +from mem0 import Memory + +config = { + "vector_store": { + "provider": "chroma", + "config": { + "collection_name": "my_memories", + "path": "./chroma_db" + } + }, + "llm": { + "provider": "openai", + "config": { + "model": "gpt-4o-mini" + } + }, + "rerank": { + "provider": "zero_entropy", + "config": { + "model": "zerank-1", # or "zerank-1-small" + "api_key": "your-zero-entropy-api-key", # or set ZERO_ENTROPY_API_KEY + "top_k": 5 + } + } +} + +memory = Memory.from_config(config) +``` + +## Environment Variables + +Set your API key as an environment variable: + +```bash +export ZERO_ENTROPY_API_KEY="your-api-key" +``` + +## Usage Example + +```python Python +import os +from mem0 import Memory + +# Set API key +os.environ["ZERO_ENTROPY_API_KEY"] = "your-api-key" + +# Initialize memory with Zero Entropy reranker +config = { + "vector_store": {"provider": "chroma"}, + "llm": {"provider": "openai", "config": {"model": "gpt-4o-mini"}}, + "rerank": {"provider": "zero_entropy", "config": {"model": "zerank-1"}} +} + +memory = Memory.from_config(config) + +# Add memories +messages = [ + {"role": "user", "content": "I love Italian pasta, especially carbonara"}, + {"role": "user", "content": "Japanese sushi is also amazing"}, + {"role": "user", "content": "I enjoy cooking Mediterranean dishes"} +] + +memory.add(messages, user_id="alice") + +# Search with reranking +results = memory.search("What Italian food does the user like?", user_id="alice") + +for result in results['results']: + print(f"Memory: {result['memory']}") + print(f"Vector Score: {result['score']:.3f}") + print(f"Rerank Score: {result['rerank_score']:.3f}") + print() +``` + +## Configuration Parameters + +| Parameter | Description | Type | Default | +|-----------|-------------|------|---------| +| `model` | Model to use: `"zerank-1"` or `"zerank-1-small"` | `str` | `"zerank-1"` | +| `api_key` | Zero Entropy API key | `str` | `None` | +| `top_k` | Maximum documents to return after reranking | `int` | `None` | + +## Performance + +- **Fast**: Optimized neural architecture for low latency +- **Accurate**: State-of-the-art relevance scoring +- **Cost-effective**: ~$0.025/1M tokens processed + +## Best Practices + +1. **Model Selection**: Use `zerank-1` for best quality, `zerank-1-small` for faster processing +2. **Batch Size**: Process multiple queries together when possible +3. **Top-k Limiting**: Set reasonable `top_k` values (5-20) for best performance +4. **API Key Management**: Use environment variables for secure key storage \ No newline at end of file diff --git a/docs/components/rerankers/overview.mdx b/docs/components/rerankers/overview.mdx new file mode 100644 index 000000000..b585c5234 --- /dev/null +++ b/docs/components/rerankers/overview.mdx @@ -0,0 +1,47 @@ +--- +title: Overview +icon: "arrow-up-arrow-down" +iconType: "solid" +--- + +Mem0 includes built-in support for various reranking providers to improve the relevance of memory search results. Rerankers post-process initial vector search results by re-scoring and re-ordering them using more sophisticated relevance models. + +## Usage + +To use a reranker, you must provide a `rerank` configuration section in your memory config. If no reranker is configured, search results will rely on vector similarity scoring alone. + +For comprehensive configuration parameters for each reranker, please refer to [Config](./config). + +## How Reranking Works + +1. **Initial Search**: Vector similarity search retrieves candidate memories +2. **Reranking**: Selected reranker re-scores candidates using advanced models +3. **Final Results**: Re-ordered results with both vector and rerank scores + + +Reranking operates as a post-processing step and can significantly improve search relevance at the cost of additional latency and API calls. + + +## Supported Rerankers + +See the list of supported rerankers below. + + + + + + + + +## When to Use Reranking + +- **Improved Relevance**: When vector search alone doesn't provide sufficiently relevant results +- **Domain-Specific Queries**: For specialized terminology or context that benefits from advanced models +- **Quality vs Speed Trade-off**: When you can accept higher latency for better search quality +- **Production Systems**: Where search quality directly impacts user experience + +Choose the reranker that best fits your use case: +- **Zero Entropy**: Best balance of speed and quality for general use +- **Cohere**: Enterprise-grade with excellent multilingual support +- **Sentence Transformer**: Local deployment for privacy-sensitive applications +- **LLM-based**: Maximum customization with custom prompts and logic \ No newline at end of file diff --git a/docs/open-source/features/overview.mdx b/docs/open-source/features/overview.mdx index acb4d9a3e..59cbe7530 100644 --- a/docs/open-source/features/overview.mdx +++ b/docs/open-source/features/overview.mdx @@ -26,6 +26,7 @@ Mem0 open-source provides a powerful, flexible foundation for AI memory manageme ### Memory Management - **Synchronous & Asynchronous Operations**: Choose between sync and async memory operations based on your application needs - **Smart Memory Retrieval**: Intelligent search and retrieval with semantic understanding +- **Advanced Reranking**: Improve search relevance with Zero Entropy, LLM-based, or custom reranking models - **Memory Persistence**: Long-term storage with automatic optimization and cleanup ### Advanced Organization diff --git a/docs/open-source/features/reranking.mdx b/docs/open-source/features/reranking.mdx new file mode 100644 index 000000000..b794604c4 --- /dev/null +++ b/docs/open-source/features/reranking.mdx @@ -0,0 +1,130 @@ +--- +title: Reranking +description: 'Improve memory search relevance with advanced reranking capabilities' +icon: "arrow-up-arrow-down" +iconType: "solid" +--- + +## Overview + +Reranking is an advanced feature that improves the relevance of memory search results by re-ordering them based on more sophisticated relevance scoring. After initial vector similarity search, rerankers use specialized models to provide more accurate relevance scores. + + +Reranking operates as a post-processing step after the initial vector search. It takes the top results from vector similarity search and re-scores them using more advanced models or custom logic. + + +## How It Works + +1. **Vector Search**: Initial semantic similarity search retrieves candidate memories +2. **Reranking**: Selected reranker re-scores candidates using advanced models +3. **Final Results**: Re-ordered results with both vector and rerank scores + +## Quick Start + +Enable reranking by adding a `rerank` section to your memory configuration: + +```python Python +from mem0 import Memory + +config = { + "vector_store": { + "provider": "chroma", + "config": { + "collection_name": "my_memories", + "path": "./chroma_db" + } + }, + "llm": { + "provider": "openai", + "config": { + "model": "gpt-4o-mini" + } + }, + "rerank": { + "provider": "zero_entropy", + "config": { + "model": "zerank-1", + "top_k": 5 + } + } +} + +memory = Memory.from_config(config) + +# Add memories +messages = [ + {"role": "user", "content": "I love Italian pasta, especially carbonara"}, + {"role": "assistant", "content": "Carbonara is a classic Roman dish!"} +] + +memory.add(messages, user_id="alice") + +# Search with reranking - results automatically include rerank scores +results = memory.search("What Italian dishes does the user like?", user_id="alice") + +for result in results['results']: + print(f"Memory: {result['memory']}") + print(f"Vector Score: {result['score']:.3f}") + print(f"Rerank Score: {result['rerank_score']:.3f}") +``` + +## Supported Providers + +Mem0 supports multiple reranking providers: + +- **[Zero Entropy](../../components/rerankers/models/zero_entropy)**: State-of-the-art neural reranking +- **[Cohere](../../components/rerankers/models/cohere)**: Enterprise-grade with multilingual support +- **[Sentence Transformer](../../components/rerankers/models/sentence_transformer)**: Local HuggingFace models +- **[LLM-based](../../components/rerankers/models/llm)**: Custom scoring using any LLM + +## When to Use Reranking + +Reranking is particularly effective for: + +- **Improved Relevance**: When vector search alone doesn't provide sufficiently relevant results +- **Domain-Specific Queries**: Specialized terminology or context that benefits from advanced models +- **Customer Support**: Finding the most relevant help articles and documentation +- **Knowledge Management**: Better search results in internal knowledge bases +- **Personal AI Assistants**: More accurate memory recall for user queries + +## Configuration Options + +Each reranker has specific configuration options. See the [Rerankers Documentation](../../components/rerankers/overview) for detailed configuration parameters. + +### Basic Configuration + +```python Python +"rerank": { + "provider": "zero_entropy", # or "cohere", "sentence_transformer", "llm" + "config": { + "top_k": 5, # Limit results after reranking + "api_key": "your-key" # Provider-specific API key + } +} +``` + +### Controlling Reranking + +You can enable or disable reranking per search: + +```python Python +# Search with reranking (default when configured) +results = memory.search("query", user_id="alice", rerank=True) + +# Search without reranking +results = memory.search("query", user_id="alice", rerank=False) +``` + +## Performance Considerations + +- **Latency**: Reranking adds processing time but significantly improves relevance +- **Cost**: API-based rerankers (Zero Entropy, Cohere, LLM) have per-request costs +- **Local Options**: Sentence Transformer reranker runs locally with no API costs +- **Quality vs Speed**: Balance based on your application's requirements + +## Next Steps + +- Explore specific [reranker providers](../../components/rerankers/overview) and their capabilities +- Learn about [configuration options](../../components/rerankers/config) for fine-tuning +- Check out [Vector Stores](../../components/vectordbs/overview) for different storage backends +- See [Async Memory](./async-memory) for non-blocking reranking operations \ No newline at end of file diff --git a/mem0/configs/rerankers/llm.py b/mem0/configs/rerankers/llm.py new file mode 100644 index 000000000..414dd0a56 --- /dev/null +++ b/mem0/configs/rerankers/llm.py @@ -0,0 +1,48 @@ +from typing import Optional +from pydantic import Field + +from mem0.configs.rerankers.base import BaseRerankerConfig + + +class LLMRerankerConfig(BaseRerankerConfig): + """ + Configuration for LLM-based reranker. + + Attributes: + model (str): LLM model to use for reranking. Defaults to "gpt-4o-mini". + api_key (str): API key for the LLM provider. + provider (str): LLM provider. Defaults to "openai". + top_k (int): Number of top documents to return after reranking. + temperature (float): Temperature for LLM generation. Defaults to 0.0 for deterministic scoring. + max_tokens (int): Maximum tokens for LLM response. Defaults to 100. + scoring_prompt (str): Custom prompt template for scoring documents. + """ + + model: str = Field( + default="gpt-4o-mini", + description="LLM model to use for reranking" + ) + api_key: Optional[str] = Field( + default=None, + description="API key for the LLM provider" + ) + provider: str = Field( + default="openai", + description="LLM provider (openai, anthropic, etc.)" + ) + top_k: Optional[int] = Field( + default=None, + description="Number of top documents to return after reranking" + ) + temperature: float = Field( + default=0.0, + description="Temperature for LLM generation" + ) + max_tokens: int = Field( + default=100, + description="Maximum tokens for LLM response" + ) + scoring_prompt: Optional[str] = Field( + default=None, + description="Custom prompt template for scoring documents" + ) \ No newline at end of file diff --git a/mem0/configs/rerankers/zero_entropy.py b/mem0/configs/rerankers/zero_entropy.py new file mode 100644 index 000000000..197a41d59 --- /dev/null +++ b/mem0/configs/rerankers/zero_entropy.py @@ -0,0 +1,28 @@ +from typing import Optional +from pydantic import Field + +from mem0.configs.rerankers.base import BaseRerankerConfig + + +class ZeroEntropyRerankerConfig(BaseRerankerConfig): + """ + Configuration for Zero Entropy reranker. + + Attributes: + model (str): Model to use for reranking. Defaults to "zerank-1". + api_key (str): Zero Entropy API key. If not provided, will try to read from ZERO_ENTROPY_API_KEY environment variable. + top_k (int): Number of top documents to return after reranking. + """ + + model: str = Field( + default="zerank-1", + description="Model to use for reranking. Available models: zerank-1, zerank-1-small" + ) + api_key: Optional[str] = Field( + default=None, + description="Zero Entropy API key" + ) + top_k: Optional[int] = Field( + default=None, + description="Number of top documents to return after reranking" + ) \ No newline at end of file diff --git a/mem0/memory/main.py b/mem0/memory/main.py index c7a7c0da4..6ad9e6418 100644 --- a/mem0/memory/main.py +++ b/mem0/memory/main.py @@ -637,18 +637,22 @@ class Memory(MemoryBase): limit (int, optional): Limit the number of results. Defaults to 100. filters (dict, optional): Legacy filters to apply to the search. Defaults to None. threshold (float, optional): Minimum score for a memory to be included in the results. Defaults to None. - metadata_filters (dict, optional): Enhanced metadata filtering with operators: + filters (dict, optional): Enhanced metadata filtering with operators: - {"key": "value"} - exact match - - {"key": {"$eq": "value"}} - equals - - {"key": {"$ne": "value"}} - not equals - - {"key": {"$in": ["val1", "val2"]}} - in list - - {"key": {"$nin": ["val1", "val2"]}} - not in list - - {"key": {"$gt": 10}} - greater than - - {"key": {"$gte": 10}} - greater than or equal - - {"key": {"$lt": 10}} - less than - - {"key": {"$lte": 10}} - less than or equal - - {"$and": [filter1, filter2]} - logical AND - - {"$or": [filter1, filter2]} - logical OR + - {"key": {"eq": "value"}} - equals + - {"key": {"ne": "value"}} - not equals + - {"key": {"in": ["val1", "val2"]}} - in list + - {"key": {"nin": ["val1", "val2"]}} - not in list + - {"key": {"gt": 10}} - greater than + - {"key": {"gte": 10}} - greater than or equal + - {"key": {"lt": 10}} - less than + - {"key": {"lte": 10}} - less than or equal + - {"key": {"contains": "text"}} - contains text + - {"key": {"icontains": "text"}} - case-insensitive contains + - {"key": "*"} - wildcard match (any value) + - {"AND": [filter1, filter2]} - logical AND + - {"OR": [filter1, filter2]} - logical OR + - {"NOT": [filter1]} - logical NOT Returns: dict: A dictionary containing the search results, typically under a "results" key, @@ -733,33 +737,55 @@ class Memory(MemoryBase): def process_condition(key: str, condition: Any) -> Dict[str, Any]: if not isinstance(condition, dict): # Simple equality: {"key": "value"} - return {key: {"$eq": condition}} + if condition == "*": + # Wildcard: match everything for this field (implementation depends on vector store) + return {key: "*"} + return {key: condition} result = {} for operator, value in condition.items(): - if operator in ["$eq", "$ne", "$gt", "$gte", "$lt", "$lte", "$in", "$nin"]: - result[key] = {operator: value} + # Map platform operators to universal format that can be translated by each vector store + operator_map = { + "eq": "eq", "ne": "ne", "gt": "gt", "gte": "gte", + "lt": "lt", "lte": "lte", "in": "in", "nin": "nin", + "contains": "contains", "icontains": "icontains" + } + + if operator in operator_map: + result[key] = {operator_map[operator]: value} else: raise ValueError(f"Unsupported metadata filter operator: {operator}") return result for key, value in metadata_filters.items(): - if key == "$and": + if key == "AND": # Logical AND: combine multiple conditions if not isinstance(value, list): - raise ValueError("$and operator requires a list of conditions") + raise ValueError("AND operator requires a list of conditions") for condition in value: for sub_key, sub_value in condition.items(): processed_filters.update(process_condition(sub_key, sub_value)) - elif key == "$or": - # Logical OR: for now, we'll apply the first condition - # Note: Full OR support would require vector store level implementation + elif key == "OR": + # Logical OR: Pass through to vector store for implementation-specific handling if not isinstance(value, list) or not value: - raise ValueError("$or operator requires a non-empty list of conditions") - # Apply first condition as fallback - first_condition = value[0] - for sub_key, sub_value in first_condition.items(): - processed_filters.update(process_condition(sub_key, sub_value)) + raise ValueError("OR operator requires a non-empty list of conditions") + # Store OR conditions in a way that vector stores can interpret + processed_filters["$or"] = [] + for condition in value: + or_condition = {} + for sub_key, sub_value in condition.items(): + or_condition.update(process_condition(sub_key, sub_value)) + processed_filters["$or"].append(or_condition) + elif key == "NOT": + # Logical NOT: Pass through to vector store for implementation-specific handling + if not isinstance(value, list) or not value: + raise ValueError("NOT operator requires a non-empty list of conditions") + processed_filters["$not"] = [] + for condition in value: + not_condition = {} + for sub_key, sub_value in condition.items(): + not_condition.update(process_condition(sub_key, sub_value)) + processed_filters["$not"].append(not_condition) else: processed_filters.update(process_condition(key, value)) @@ -779,14 +805,17 @@ class Memory(MemoryBase): return False for key, value in filters.items(): - # Check for logical operators - if key in ["$and", "$or"]: + # Check for platform-style logical operators + if key in ["AND", "OR", "NOT"]: return True - # Check for advanced comparison operators + # Check for comparison operators (without $ prefix for universal compatibility) if isinstance(value, dict): for op in value.keys(): - if op in ["$eq", "$ne", "$gt", "$gte", "$lt", "$lte", "$in", "$nin"]: + if op in ["eq", "ne", "gt", "gte", "lt", "lte", "in", "nin", "contains", "icontains"]: return True + # Check for wildcard values + if value == "*": + return True return False def _search_vector_store(self, query, filters, limit, threshold: Optional[float] = None): @@ -1603,18 +1632,22 @@ class AsyncMemory(MemoryBase): limit (int, optional): Limit the number of results. Defaults to 100. filters (dict, optional): Legacy filters to apply to the search. Defaults to None. threshold (float, optional): Minimum score for a memory to be included in the results. Defaults to None. - metadata_filters (dict, optional): Enhanced metadata filtering with operators: + filters (dict, optional): Enhanced metadata filtering with operators: - {"key": "value"} - exact match - - {"key": {"$eq": "value"}} - equals - - {"key": {"$ne": "value"}} - not equals - - {"key": {"$in": ["val1", "val2"]}} - in list - - {"key": {"$nin": ["val1", "val2"]}} - not in list - - {"key": {"$gt": 10}} - greater than - - {"key": {"$gte": 10}} - greater than or equal - - {"key": {"$lt": 10}} - less than - - {"key": {"$lte": 10}} - less than or equal - - {"$and": [filter1, filter2]} - logical AND - - {"$or": [filter1, filter2]} - logical OR + - {"key": {"eq": "value"}} - equals + - {"key": {"ne": "value"}} - not equals + - {"key": {"in": ["val1", "val2"]}} - in list + - {"key": {"nin": ["val1", "val2"]}} - not in list + - {"key": {"gt": 10}} - greater than + - {"key": {"gte": 10}} - greater than or equal + - {"key": {"lt": 10}} - less than + - {"key": {"lte": 10}} - less than or equal + - {"key": {"contains": "text"}} - contains text + - {"key": {"icontains": "text"}} - case-insensitive contains + - {"key": "*"} - wildcard match (any value) + - {"AND": [filter1, filter2]} - logical AND + - {"OR": [filter1, filter2]} - logical OR + - {"NOT": [filter1]} - logical NOT Returns: dict: A dictionary containing the search results, typically under a "results" key, diff --git a/mem0/reranker/llm_reranker.py b/mem0/reranker/llm_reranker.py new file mode 100644 index 000000000..6eff0dbe8 --- /dev/null +++ b/mem0/reranker/llm_reranker.py @@ -0,0 +1,127 @@ +import os +import re +from typing import List, Dict, Any, Optional + +from mem0.reranker.base import BaseReranker +from mem0.utils.factory import LlmFactory + + +class LLMReranker(BaseReranker): + """LLM-based reranker implementation.""" + + def __init__(self, config): + """ + Initialize LLM reranker. + + Args: + config: LLMRerankerConfig object with configuration parameters + """ + self.config = config + + # Create LLM configuration for the factory + llm_config = { + "model": config.model, + "temperature": config.temperature, + "max_tokens": config.max_tokens, + } + + # Add API key if provided + if config.api_key: + llm_config["api_key"] = config.api_key + + # Initialize LLM using the factory + self.llm = LlmFactory.create(config.provider, llm_config) + + # Default scoring prompt + self.scoring_prompt = config.scoring_prompt or self._get_default_prompt() + + def _get_default_prompt(self) -> str: + """Get the default scoring prompt template.""" + return """You are a relevance scoring assistant. Given a query and a document, you need to score how relevant the document is to the query. + +Score the relevance on a scale from 0.0 to 1.0, where: +- 1.0 = Perfectly relevant and directly answers the query +- 0.8-0.9 = Highly relevant with good information +- 0.6-0.7 = Moderately relevant with some useful information +- 0.4-0.5 = Slightly relevant with limited useful information +- 0.0-0.3 = Not relevant or no useful information + +Query: "{query}" +Document: "{document}" + +Provide only a single numerical score between 0.0 and 1.0. Do not include any explanation or additional text.""" + + def _extract_score(self, response_text: str) -> float: + """Extract numerical score from LLM response.""" + # Look for decimal numbers between 0.0 and 1.0 + pattern = r'\b([01](?:\.\d+)?)\b' + matches = re.findall(pattern, response_text) + + if matches: + score = float(matches[0]) + return min(max(score, 0.0), 1.0) # Clamp between 0.0 and 1.0 + + # Fallback: return 0.5 if no valid score found + return 0.5 + + def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = None) -> List[Dict[str, Any]]: + """ + Rerank documents using LLM scoring. + + Args: + query: The search query + documents: List of documents to rerank + top_k: Number of top documents to return + + Returns: + List of reranked documents with rerank_score + """ + if not documents: + return documents + + scored_docs = [] + + for doc in documents: + # Extract text content + if 'memory' in doc: + doc_text = doc['memory'] + elif 'text' in doc: + doc_text = doc['text'] + elif 'content' in doc: + doc_text = doc['content'] + else: + doc_text = str(doc) + + try: + # Generate scoring prompt + prompt = self.scoring_prompt.format(query=query, document=doc_text) + + # Get LLM response + response = self.llm.generate_response( + messages=[{"role": "user", "content": prompt}] + ) + + # Extract score from response + score = self._extract_score(response) + + # Create scored document + scored_doc = doc.copy() + scored_doc['rerank_score'] = score + scored_docs.append(scored_doc) + + except Exception as e: + # Fallback: assign neutral score if scoring fails + scored_doc = doc.copy() + scored_doc['rerank_score'] = 0.5 + scored_docs.append(scored_doc) + + # Sort by relevance score in descending order + scored_docs.sort(key=lambda x: x['rerank_score'], reverse=True) + + # Apply top_k limit + if top_k: + scored_docs = scored_docs[:top_k] + elif self.config.top_k: + scored_docs = scored_docs[:self.config.top_k] + + return scored_docs \ No newline at end of file diff --git a/mem0/reranker/zero_entropy_reranker.py b/mem0/reranker/zero_entropy_reranker.py new file mode 100644 index 000000000..0bf1fd5cb --- /dev/null +++ b/mem0/reranker/zero_entropy_reranker.py @@ -0,0 +1,96 @@ +import os +from typing import List, Dict, Any, Optional + +from mem0.reranker.base import BaseReranker + +try: + from zeroentropy import ZeroEntropy + ZERO_ENTROPY_AVAILABLE = True +except ImportError: + ZERO_ENTROPY_AVAILABLE = False + + +class ZeroEntropyReranker(BaseReranker): + """Zero Entropy-based reranker implementation.""" + + def __init__(self, config): + """ + Initialize Zero Entropy reranker. + + Args: + config: ZeroEntropyRerankerConfig object with configuration parameters + """ + if not ZERO_ENTROPY_AVAILABLE: + raise ImportError("zeroentropy package is required for ZeroEntropyReranker. Install with: pip install zeroentropy") + + self.config = config + self.api_key = config.api_key or os.getenv("ZERO_ENTROPY_API_KEY") + if not self.api_key: + raise ValueError("Zero Entropy API key is required. Set ZERO_ENTROPY_API_KEY environment variable or pass api_key in config.") + + self.model = config.model or "zerank-1" + + # Initialize Zero Entropy client + if self.api_key: + self.client = ZeroEntropy(api_key=self.api_key) + else: + self.client = ZeroEntropy() # Will use ZERO_ENTROPY_API_KEY from environment + + def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = None) -> List[Dict[str, Any]]: + """ + Rerank documents using Zero Entropy's rerank API. + + Args: + query: The search query + documents: List of documents to rerank + top_k: Number of top documents to return + + Returns: + List of reranked documents with rerank_score + """ + if not documents: + return documents + + # Extract text content for reranking + doc_texts = [] + for doc in documents: + if 'memory' in doc: + doc_texts.append(doc['memory']) + elif 'text' in doc: + doc_texts.append(doc['text']) + elif 'content' in doc: + doc_texts.append(doc['content']) + else: + doc_texts.append(str(doc)) + + try: + # Call Zero Entropy rerank API + response = self.client.models.rerank( + model=self.model, + query=query, + documents=doc_texts, + ) + + # Create reranked results + reranked_docs = [] + for result in response.results: + original_doc = documents[result.index].copy() + original_doc['rerank_score'] = result.relevance_score + reranked_docs.append(original_doc) + + # Sort by relevance score in descending order + reranked_docs.sort(key=lambda x: x['rerank_score'], reverse=True) + + # Apply top_k limit + if top_k: + reranked_docs = reranked_docs[:top_k] + elif self.config.top_k: + reranked_docs = reranked_docs[:self.config.top_k] + + return reranked_docs + + except Exception as e: + # Fallback to original order if reranking fails + for doc in documents: + doc['rerank_score'] = 0.0 + return documents[:top_k] if top_k else documents \ No newline at end of file diff --git a/mem0/utils/factory.py b/mem0/utils/factory.py index e3a453622..7d33119c7 100644 --- a/mem0/utils/factory.py +++ b/mem0/utils/factory.py @@ -13,6 +13,8 @@ from mem0.configs.llms.vllm import VllmConfig from mem0.configs.rerankers.base import BaseRerankerConfig from mem0.configs.rerankers.cohere import CohereRerankerConfig from mem0.configs.rerankers.sentence_transformer import SentenceTransformerRerankerConfig +from mem0.configs.rerankers.zero_entropy import ZeroEntropyRerankerConfig +from mem0.configs.rerankers.llm import LLMRerankerConfig from mem0.embeddings.mock import MockEmbeddings @@ -229,6 +231,8 @@ class RerankerFactory: provider_to_class = { "cohere": ("mem0.reranker.cohere_reranker.CohereReranker", CohereRerankerConfig), "sentence_transformer": ("mem0.reranker.sentence_transformer_reranker.SentenceTransformerReranker", SentenceTransformerRerankerConfig), + "zero_entropy": ("mem0.reranker.zero_entropy_reranker.ZeroEntropyReranker", ZeroEntropyRerankerConfig), + "llm": ("mem0.reranker.llm_reranker.LLMReranker", LLMRerankerConfig), } @classmethod diff --git a/mem0/vector_stores/chroma.py b/mem0/vector_stores/chroma.py index 681d4626c..62c802ad0 100644 --- a/mem0/vector_stores/chroma.py +++ b/mem0/vector_stores/chroma.py @@ -241,14 +241,79 @@ class ChromaDB(VectorStoreBase): Returns: dict[str, any]: Properly formatted where clause for ChromaDB. """ - # If only one filter is supplied, return it as is - # (no need to wrap in $and based on chroma docs) if where is None: return {} - if len(where.keys()) <= 1: - return where - where_filters = [] - for k, v in where.items(): - if isinstance(v, str): - where_filters.append({k: v}) - return {"$and": where_filters} + + def convert_condition(key: str, value: any) -> dict: + """Convert universal filter format to ChromaDB format.""" + if value == "*": + # Wildcard - match any value (ChromaDB doesn't have direct wildcard, so we skip this filter) + return None + elif isinstance(value, dict): + # Handle comparison operators + chroma_condition = {} + for op, val in value.items(): + if op == "eq": + chroma_condition[key] = {"$eq": val} + elif op == "ne": + chroma_condition[key] = {"$ne": val} + elif op == "gt": + chroma_condition[key] = {"$gt": val} + elif op == "gte": + chroma_condition[key] = {"$gte": val} + elif op == "lt": + chroma_condition[key] = {"$lt": val} + elif op == "lte": + chroma_condition[key] = {"$lte": val} + elif op == "in": + chroma_condition[key] = {"$in": val} + elif op == "nin": + chroma_condition[key] = {"$nin": val} + elif op in ["contains", "icontains"]: + # ChromaDB doesn't support contains, fallback to equality + chroma_condition[key] = {"$eq": val} + else: + # Unknown operator, treat as equality + chroma_condition[key] = {"$eq": val} + return chroma_condition + else: + # Simple equality + return {key: {"$eq": value}} + + processed_filters = [] + + for key, value in where.items(): + if key == "$or": + # Handle OR conditions + or_conditions = [] + for condition in value: + or_condition = {} + for sub_key, sub_value in condition.items(): + converted = convert_condition(sub_key, sub_value) + if converted: + or_condition.update(converted) + if or_condition: + or_conditions.append(or_condition) + + if len(or_conditions) > 1: + processed_filters.append({"$or": or_conditions}) + elif len(or_conditions) == 1: + processed_filters.append(or_conditions[0]) + + elif key == "$not": + # Handle NOT conditions - ChromaDB doesn't have direct NOT, so we'll skip for now + continue + + else: + # Regular condition + converted = convert_condition(key, value) + if converted: + processed_filters.append(converted) + + # Return appropriate format based on number of conditions + if len(processed_filters) == 0: + return {} + elif len(processed_filters) == 1: + return processed_filters[0] + else: + return {"$and": processed_filters}