Compare commits
40 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| ea09b5f7f0 | |||
| 5258fd91ea | |||
| b305d674de | |||
| 7c24601d0f | |||
| 50c0285cb2 | |||
| 0a78198bb5 | |||
| edaeb78ccf | |||
| f80be2d2ea | |||
| 8700165b42 | |||
| 18fb92f1f8 | |||
| 14fc6bbadd | |||
| 5070a1d83e | |||
| 8a9088ea9d | |||
| 48b24f6f12 | |||
| f6ddd5ffc5 | |||
| b43a116b3c | |||
| 50512a5f03 | |||
| e3e107b31d | |||
| 21a04541ea | |||
| cdd5d8ac76 | |||
| 11094f504e | |||
| 5acaae5f56 | |||
| 4547d870af | |||
| dc0d8e0932 | |||
| c558eae9ce | |||
| abb9af66a6 | |||
| 4800e0344c | |||
| 439b425c61 | |||
| 2855f1635b | |||
| 08b67b4a78 | |||
| 1bddd46ed2 | |||
| 6ecdadfd97 | |||
| 4119040005 | |||
| 873eef6ef8 | |||
| 445fed4d3f | |||
| 52fd3e0dd4 | |||
| 8fd0e1f3b0 | |||
| 11fc4a8451 | |||
| e22293294e | |||
| 73e53aaff1 |
@@ -67,6 +67,10 @@ We use `pytest` to test our code. You can run the tests by running the following
|
||||
poetry run pytest
|
||||
```
|
||||
|
||||
|
||||
Several packages have been removed from Poetry to make the package lighter. Therefore, it is recommended to run `make install_all` to install the remaining packages and ensure all tests pass.
|
||||
|
||||
|
||||
Make sure that all tests pass before submitting a pull request.
|
||||
|
||||
## 🚀 Release Process
|
||||
|
||||
@@ -11,7 +11,7 @@ install:
|
||||
|
||||
install_all:
|
||||
poetry install --all-extras
|
||||
poetry run pip install pinecone-text pinecone-client langchain-anthropic "unstructured[local-inference, all-docs]"
|
||||
poetry run pip install pinecone-text pinecone-client langchain-anthropic "unstructured[local-inference, all-docs]" ollama deepgram-sdk==3.2.7 langchain-huggingface psutil
|
||||
|
||||
install_es:
|
||||
poetry install --extras elasticsearch
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
llm:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/mistralai/completion/models/mistral-7B-Instruct"
|
||||
model_kwargs:
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
|
||||
embedder:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/clarifai/main/models/BAAI-bge-base-en-v15"
|
||||
@@ -26,6 +26,11 @@ llm:
|
||||
top_p: 1
|
||||
stream: false
|
||||
api_key: sk-xxx
|
||||
model_kwargs:
|
||||
response_format:
|
||||
type: json_object
|
||||
api_version: 2024-02-01
|
||||
http_client_proxies: http://testproxy.mem0.net:8000
|
||||
prompt: |
|
||||
Use the following pieces of context to answer the query at the end.
|
||||
If you don't know the answer, just say that you don't know, don't try to make up an answer.
|
||||
@@ -83,7 +88,10 @@ cache:
|
||||
"stream": false,
|
||||
"prompt": "Use the following pieces of context to answer the query at the end.\nIf you don't know the answer, just say that you don't know, don't try to make up an answer.\n$context\n\nQuery: $query\n\nHelpful Answer:",
|
||||
"system_prompt": "Act as William Shakespeare. Answer the following questions in the style of William Shakespeare.",
|
||||
"api_key": "sk-xxx"
|
||||
"api_key": "sk-xxx",
|
||||
"model_kwargs": {"response_format": {"type": "json_object"}},
|
||||
"api_version": "2024-02-01",
|
||||
"http_client_proxies": "http://testproxy.mem0.net:8000",
|
||||
}
|
||||
},
|
||||
"vectordb": {
|
||||
@@ -143,7 +151,9 @@ config = {
|
||||
'system_prompt': (
|
||||
"Act as William Shakespeare. Answer the following questions in the style of William Shakespeare."
|
||||
),
|
||||
'api_key': 'sk-xxx'
|
||||
'api_key': 'sk-xxx',
|
||||
"model_kwargs": {"response_format": {"type": "json_object"}},
|
||||
"http_client_proxies": "http://testproxy.mem0.net:8000",
|
||||
}
|
||||
},
|
||||
'vectordb': {
|
||||
@@ -204,22 +214,27 @@ Alright, let's dive into what each key means in the yaml config above:
|
||||
- `number_documents` (Integer): Number of documents to pull from the vectordb as context, defaults to 1
|
||||
- `api_key` (String): The API key for the language model.
|
||||
- `model_kwargs` (Dict): Keyword arguments to pass to the language model. Used for `aws_bedrock` provider, since it requires different arguments for each model.
|
||||
- `http_client_proxies` (Dict | String): The proxy server settings used to create `self.http_client` using `httpx.Client(proxies=http_client_proxies)`
|
||||
- `http_async_client_proxies` (Dict | String): The proxy server settings for async calls used to create `self.http_async_client` using `httpx.AsyncClient(proxies=http_async_client_proxies)`
|
||||
3. `vectordb` Section:
|
||||
- `provider` (String): The provider for the vector database, set to 'chroma'. You can find the full list of vector database providers in [our docs](/components/vector-databases).
|
||||
- `config`:
|
||||
- `collection_name` (String): The initial collection name for the vectordb, set to 'full-stack-app'.
|
||||
- `dir` (String): The directory for the local database, set to 'db'.
|
||||
- `allow_reset` (Boolean): Indicates whether resetting the vectordb is allowed, set to true.
|
||||
- `batch_size` (Integer): The batch size for docs insertion in vectordb, defaults to `100`
|
||||
<Note>We recommend you to checkout vectordb specific config [here](https://docs.embedchain.ai/components/vector-databases)</Note>
|
||||
4. `embedder` Section:
|
||||
- `provider` (String): The provider for the embedder, set to 'openai'. You can find the full list of embedding model providers in [our docs](/components/embedding-models).
|
||||
- `config`:
|
||||
- `model` (String): The specific model used for text embedding, 'text-embedding-ada-002'.
|
||||
- `vector_dimension` (Integer): The vector dimension of the embedding model. [Defaults](https://github.com/embedchain/embedchain/blob/e572b5a3dc1b66f1e9b3357d11a88c63b5ce06e3/embedchain/models/vector_dimensions.py)
|
||||
- `vector_dimension` (Integer): The vector dimension of the embedding model. [Defaults](https://github.com/embedchain/embedchain/blob/main/embedchain/models/vector_dimensions.py)
|
||||
- `api_key` (String): The API key for the embedding model.
|
||||
- `endpoint` (String): The endpoint for the HuggingFace embedding model.
|
||||
- `deployment_name` (String): The deployment name for the embedding model.
|
||||
- `title` (String): The title for the embedding model for Google Embedder.
|
||||
- `task_type` (String): The task type for the embedding model for Google Embedder.
|
||||
- `model_kwargs` (Dict): Used to pass extra arguments to embedders.
|
||||
5. `chunker` Section:
|
||||
- `chunk_size` (Integer): The size of each chunk of text that is sent to the language model.
|
||||
- `chunk_overlap` (Integer): The amount of overlap between each chunk of text.
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
---
|
||||
title: "🎤 Audio"
|
||||
---
|
||||
|
||||
|
||||
To use an audio as data source, just add `data_type` as `audio` and pass in the path of the audio (local or hosted).
|
||||
|
||||
We use [Deepgram](https://developers.deepgram.com/docs/introduction) to transcribe the audiot to text, and then use the generated text as the data source.
|
||||
|
||||
You would require an Deepgram API key which is available [here](https://console.deepgram.com/signup?jump=keys) to use this feature.
|
||||
|
||||
### Without customization
|
||||
|
||||
```python
|
||||
import os
|
||||
from embedchain import App
|
||||
|
||||
os.environ["DEEPGRAM_API_KEY"] = "153xxx"
|
||||
|
||||
app = App()
|
||||
app.add("introduction.wav", data_type="audio")
|
||||
response = app.query("What is my name and how old am I?")
|
||||
print(response)
|
||||
# Answer: Your name is Dave and you are 21 years old.
|
||||
```
|
||||
@@ -9,6 +9,7 @@ Embedchain comes with built-in support for various data sources. We handle the c
|
||||
<Card title="CSV file" href="/components/data-sources/csv"></Card>
|
||||
<Card title="JSON file" href="/components/data-sources/json"></Card>
|
||||
<Card title="Text" href="/components/data-sources/text"></Card>
|
||||
<Card title="Text File" href="/components/data-sources/text-file"></Card>
|
||||
<Card title="Directory" href="/components/data-sources/directory"></Card>
|
||||
<Card title="Web page" href="/components/data-sources/web-page"></Card>
|
||||
<Card title="Youtube Channel" href="/components/data-sources/youtube-channel"></Card>
|
||||
@@ -33,6 +34,7 @@ Embedchain comes with built-in support for various data sources. We handle the c
|
||||
<Card title="Beehiiv" href="/components/data-sources/beehiiv"></Card>
|
||||
<Card title="Dropbox" href="/components/data-sources/dropbox"></Card>
|
||||
<Card title="Image" href="/components/data-sources/image"></Card>
|
||||
<Card title="Audio" href="/components/data-sources/audio"></Card>
|
||||
<Card title="Custom" href="/components/data-sources/custom"></Card>
|
||||
</CardGroup>
|
||||
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
---
|
||||
title: '📄 Text file'
|
||||
---
|
||||
|
||||
To add a .txt file, specify the data_type as `text_file`. The URL provided in the first parameter of the `add` function, should be a local path. Eg:
|
||||
|
||||
```python
|
||||
from embedchain import App
|
||||
|
||||
app = App()
|
||||
app.add('path/to/file.txt', data_type="text_file")
|
||||
|
||||
app.query("Summarize the information of the text file")
|
||||
```
|
||||
@@ -16,6 +16,7 @@ Embedchain supports several embedding models from the following providers:
|
||||
<Card title="NVIDIA AI" href="#nvidia-ai"></Card>
|
||||
<Card title="Cohere" href="#cohere"></Card>
|
||||
<Card title="Ollama" href="#ollama"></Card>
|
||||
<Card title="Clarifai" href="#clarifai"></Card>
|
||||
</CardGroup>
|
||||
|
||||
## OpenAI
|
||||
@@ -191,6 +192,8 @@ embedder:
|
||||
provider: huggingface
|
||||
config:
|
||||
model: 'sentence-transformers/all-mpnet-base-v2'
|
||||
model_kwargs:
|
||||
trust_remote_code: True # Only use if you trust your embedder
|
||||
```
|
||||
|
||||
</CodeGroup>
|
||||
@@ -385,4 +388,51 @@ embedder:
|
||||
model: 'all-minilm:latest'
|
||||
```
|
||||
|
||||
</CodeGroup>
|
||||
|
||||
## Clarifai
|
||||
|
||||
Install related dependencies using the following command:
|
||||
|
||||
```bash
|
||||
pip install --upgrade 'embedchain[clarifai]'
|
||||
```
|
||||
|
||||
set the `CLARIFAI_PAT` as environment variable which you can find in the [security page](https://clarifai.com/settings/security). Optionally you can also pass the PAT key as parameters in LLM/Embedder class.
|
||||
|
||||
Now you are all set with exploring Embedchain.
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python main.py
|
||||
import os
|
||||
from embedchain import App
|
||||
|
||||
os.environ["CLARIFAI_PAT"] = "XXX"
|
||||
|
||||
# load llm and embedder configuration from config.yaml file
|
||||
app = App.from_config(config_path="config.yaml")
|
||||
|
||||
#Now let's add some data.
|
||||
app.add("https://www.forbes.com/profile/elon-musk")
|
||||
|
||||
#Query the app
|
||||
response = app.query("what college degrees does elon musk have?")
|
||||
```
|
||||
Head to [Clarifai Platform](https://clarifai.com/explore/models?page=1&perPage=24&filterData=%5B%7B%22field%22%3A%22output_fields%22%2C%22value%22%3A%5B%22embeddings%22%5D%7D%5D) to explore all the State of the Art embedding models available to use.
|
||||
For passing LLM model inference parameters use `model_kwargs` argument in the config file. Also you can use `api_key` argument to pass `CLARIFAI_PAT` in the config.
|
||||
|
||||
```yaml config.yaml
|
||||
llm:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/mistralai/completion/models/mistral-7B-Instruct"
|
||||
model_kwargs:
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
embedder:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/clarifai/main/models/BAAI-bge-base-en-v15"
|
||||
```
|
||||
</CodeGroup>
|
||||
@@ -15,6 +15,7 @@ Embedchain comes with built-in support for various popular large language models
|
||||
<Card title="Together" href="#together"></Card>
|
||||
<Card title="Ollama" href="#ollama"></Card>
|
||||
<Card title="vLLM" href="#vllm"></Card>
|
||||
<Card title="Clarifai" href="#clarifai"></Card>
|
||||
<Card title="GPT4All" href="#gpt4all"></Card>
|
||||
<Card title="JinaChat" href="#jinachat"></Card>
|
||||
<Card title="Hugging Face" href="#hugging-face"></Card>
|
||||
@@ -193,8 +194,8 @@ import os
|
||||
from embedchain import App
|
||||
|
||||
os.environ["OPENAI_API_TYPE"] = "azure"
|
||||
os.environ["OPENAI_API_BASE"] = "https://xxx.openai.azure.com/"
|
||||
os.environ["OPENAI_API_KEY"] = "xxx"
|
||||
os.environ["AZURE_OPENAI_ENDPOINT"] = "https://xxx.openai.azure.com/"
|
||||
os.environ["AZURE_OPENAI_KEY"] = "xxx"
|
||||
os.environ["OPENAI_API_VERSION"] = "xxx"
|
||||
|
||||
app = App.from_config(config_path="config.yaml")
|
||||
@@ -330,6 +331,7 @@ Setup Ollama using https://github.com/jmorganca/ollama
|
||||
|
||||
```python main.py
|
||||
import os
|
||||
os.environ["OLLAMA_HOST"] = "http://127.0.0.1:11434"
|
||||
from embedchain import App
|
||||
|
||||
# load llm configuration from config.yaml file
|
||||
@@ -345,6 +347,12 @@ llm:
|
||||
top_p: 1
|
||||
stream: true
|
||||
base_url: 'http://localhost:11434'
|
||||
embedder:
|
||||
provider: ollama
|
||||
config:
|
||||
model: znbang/bge:small-en-v1.5-q8_0
|
||||
base_url: http://localhost:11434
|
||||
|
||||
```
|
||||
|
||||
</CodeGroup>
|
||||
@@ -378,6 +386,54 @@ llm:
|
||||
|
||||
</CodeGroup>
|
||||
|
||||
## Clarifai
|
||||
|
||||
Install related dependencies using the following command:
|
||||
|
||||
```bash
|
||||
pip install --upgrade 'embedchain[clarifai]'
|
||||
```
|
||||
|
||||
set the `CLARIFAI_PAT` as environment variable which you can find in the [security page](https://clarifai.com/settings/security). Optionally you can also pass the PAT key as parameters in LLM/Embedder class.
|
||||
|
||||
Now you are all set with exploring Embedchain.
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python main.py
|
||||
import os
|
||||
from embedchain import App
|
||||
|
||||
os.environ["CLARIFAI_PAT"] = "XXX"
|
||||
|
||||
# load llm configuration from config.yaml file
|
||||
app = App.from_config(config_path="config.yaml")
|
||||
|
||||
#Now let's add some data.
|
||||
app.add("https://www.forbes.com/profile/elon-musk")
|
||||
|
||||
#Query the app
|
||||
response = app.query("what college degrees does elon musk have?")
|
||||
```
|
||||
Head to [Clarifai Platform](https://clarifai.com/explore/models?page=1&perPage=24&filterData=%5B%7B%22field%22%3A%22use_cases%22%2C%22value%22%3A%5B%22llm%22%5D%7D%5D) to browse various State-of-the-Art LLM models for your use case.
|
||||
For passing model inference parameters use `model_kwargs` argument in the config file. Also you can use `api_key` argument to pass `CLARIFAI_PAT` in the config.
|
||||
|
||||
```yaml config.yaml
|
||||
llm:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/mistralai/completion/models/mistral-7B-Instruct"
|
||||
model_kwargs:
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
embedder:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/clarifai/main/models/BAAI-bge-base-en-v15"
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
|
||||
## GPT4ALL
|
||||
|
||||
Install related dependencies using the following command:
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
---
|
||||
title: LanceDB
|
||||
---
|
||||
|
||||
## Install Embedchain with LanceDB
|
||||
|
||||
Install Embedchain, LanceDB and related dependencies using the following command:
|
||||
|
||||
```bash
|
||||
pip install "embedchain[lancedb]"
|
||||
```
|
||||
|
||||
LanceDB is a developer-friendly, open source database for AI. From hyper scalable vector search and advanced retrieval for RAG, to streaming training data and interactive exploration of large scale AI datasets.
|
||||
In order to use LanceDB as vector database, not need to set any key for local use.
|
||||
|
||||
### With OPENAI
|
||||
<CodeGroup>
|
||||
|
||||
```python main.py
|
||||
import os
|
||||
from embedchain import App
|
||||
|
||||
# set OPENAI_API_KEY as env variable
|
||||
os.environ["OPENAI_API_KEY"] = "sk-xxx"
|
||||
|
||||
# create Embedchain App and set config
|
||||
app = App.from_config(config={
|
||||
"vectordb": {
|
||||
"provider": "lancedb",
|
||||
"config": {
|
||||
"collection_name": "lancedb-index"
|
||||
}
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
# add data source and start query in
|
||||
app.add("https://www.forbes.com/profile/elon-musk")
|
||||
|
||||
# query continuously
|
||||
while(True):
|
||||
question = input("Enter question: ")
|
||||
if question in ['q', 'exit', 'quit']:
|
||||
break
|
||||
answer = app.query(question)
|
||||
print(answer)
|
||||
```
|
||||
|
||||
</CodeGroup>
|
||||
|
||||
### With Local LLM
|
||||
<CodeGroup>
|
||||
|
||||
```python main.py
|
||||
from embedchain import Pipeline as App
|
||||
|
||||
# config for Embedchain App
|
||||
config = {
|
||||
'llm': {
|
||||
'provider': 'huggingface',
|
||||
'config': {
|
||||
'model': 'mistralai/Mistral-7B-v0.1',
|
||||
'temperature': 0.1,
|
||||
'max_tokens': 250,
|
||||
'top_p': 0.1,
|
||||
'stream': True
|
||||
}
|
||||
},
|
||||
'embedder': {
|
||||
'provider': 'huggingface',
|
||||
'config': {
|
||||
'model': 'sentence-transformers/all-mpnet-base-v2'
|
||||
}
|
||||
},
|
||||
'vectordb': {
|
||||
'provider': 'lancedb',
|
||||
'config': {
|
||||
'collection_name': 'lancedb-index'
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
app = App.from_config(config=config)
|
||||
|
||||
# add data source and start query in
|
||||
app.add("https://www.tesla.com/ns_videos/2022-tesla-impact-report.pdf")
|
||||
|
||||
# query continuously
|
||||
while(True):
|
||||
question = input("Enter question: ")
|
||||
if question in ['q', 'exit', 'quit']:
|
||||
break
|
||||
answer = app.query(question)
|
||||
print(answer)
|
||||
```
|
||||
|
||||
</CodeGroup>
|
||||
|
||||
|
||||
<Snippet file="missing-vector-db-tip.mdx" />
|
||||
@@ -1,4 +0,0 @@
|
||||
---
|
||||
title: ' 🟨 Javascript'
|
||||
url: https://github.com/embedchain/embedchain/tree/main/embedchain-js
|
||||
---
|
||||
@@ -1,17 +0,0 @@
|
||||
---
|
||||
title: 'Embedchain.ai'
|
||||
description: 'Deploy your RAG application to embedchain.ai platform'
|
||||
---
|
||||
|
||||
## Deploy on Embedchain Platform
|
||||
|
||||
Embedchain enables developers to deploy their LLM-powered apps in production using the Embedchain platform. The platform offers free access to context on your data through its REST API. Once the pipeline is deployed, you can update your data sources anytime after deployment.
|
||||
|
||||
Deployment to Embedchain Platform is currently available on an invitation-only basis. To request access, please submit your information via the provided [Google Form](https://forms.gle/vigN11h7b4Ywat668). We will review your request and respond promptly.
|
||||
|
||||
|
||||
## Seeking help?
|
||||
|
||||
If you run into issues with deployment, please feel free to reach out to us via any of the following methods:
|
||||
|
||||
<Snippet file="get-help.mdx" />
|
||||
@@ -13,7 +13,6 @@ After successfully setting up and testing your RAG app locally, the next step is
|
||||
<Card title="Streamlit.io" href="/deployment/streamlit_io"></Card>
|
||||
<Card title="Gradio.app" href="/deployment/gradio_app"></Card>
|
||||
<Card title="Huggingface.co" href="/deployment/huggingface_spaces"></Card>
|
||||
<Card title="Embedchain.ai" href="/deployment/embedchain_ai"></Card>
|
||||
</CardGroup>
|
||||
|
||||
## Seeking help?
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
---
|
||||
title: '🔭 OpenLIT'
|
||||
description: 'OpenTelemetry-native Observability and Evals for LLMs & GPUs'
|
||||
---
|
||||
|
||||
Embedchain now supports integration with [OpenLIT](https://github.com/openlit/openlit).
|
||||
|
||||
## Getting Started
|
||||
|
||||
### 1. Set environment variables
|
||||
```bash
|
||||
# Setting environment variable for OpenTelemetry destination and authetication.
|
||||
export OTEL_EXPORTER_OTLP_ENDPOINT = "YOUR_OTEL_ENDPOINT"
|
||||
export OTEL_EXPORTER_OTLP_HEADERS = "YOUR_OTEL_ENDPOINT_AUTH"
|
||||
```
|
||||
|
||||
### 2. Install the OpenLIT SDK
|
||||
Open your terminal and run:
|
||||
|
||||
```shell
|
||||
pip install openlit
|
||||
```
|
||||
|
||||
### 3. Setup Your Application for Monitoring
|
||||
Now create an app using Embedchain and initialize OpenTelemetry monitoring
|
||||
|
||||
```python
|
||||
from embedchain import App
|
||||
import OpenLIT
|
||||
|
||||
# Initialize OpenLIT Auto Instrumentation for monitoring.
|
||||
openlit.init()
|
||||
|
||||
# Initialize EmbedChain application.
|
||||
app = App()
|
||||
|
||||
# Add data to your app
|
||||
app.add("https://en.wikipedia.org/wiki/Elon_Musk")
|
||||
|
||||
# Query your app
|
||||
app.query("How many companies did Elon found?")
|
||||
```
|
||||
|
||||
### 4. Visualize
|
||||
|
||||
Once you've set up data collection with OpenLIT, you can visualize and analyze this information to better understand your application's performance:
|
||||
|
||||
- **Using OpenLIT UI:** Connect to OpenLIT's UI to start exploring performance metrics. Visit the OpenLIT [Quickstart Guide](https://docs.openlit.io/latest/quickstart) for step-by-step details.
|
||||
|
||||
- **Integrate with existing Observability Tools:** If you use tools like Grafana or DataDog, you can integrate the data collected by OpenLIT. For instructions on setting up these connections, check the OpenLIT [Connections Guide](https://docs.openlit.io/latest/connections/intro).
|
||||
+4
-5
@@ -69,7 +69,8 @@
|
||||
"pages": [
|
||||
"integration/langsmith",
|
||||
"integration/chainlit",
|
||||
"integration/streamlit-mistral"
|
||||
"integration/streamlit-mistral",
|
||||
"integration/openlit"
|
||||
]
|
||||
}
|
||||
]
|
||||
@@ -155,8 +156,7 @@
|
||||
"deployment/railway",
|
||||
"deployment/streamlit_io",
|
||||
"deployment/gradio_app",
|
||||
"deployment/huggingface_spaces",
|
||||
"deployment/embedchain_ai"
|
||||
"deployment/huggingface_spaces"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -236,8 +236,7 @@
|
||||
"contribution/guidelines",
|
||||
"contribution/dev",
|
||||
"contribution/docs",
|
||||
"contribution/python",
|
||||
"contribution/javascript"
|
||||
"contribution/python"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
node_modules
|
||||
dist
|
||||
@@ -1,56 +0,0 @@
|
||||
{
|
||||
// Configuration for JavaScript files
|
||||
"extends": [
|
||||
"airbnb-base",
|
||||
"plugin:prettier/recommended"
|
||||
],
|
||||
"rules": {
|
||||
"prettier/prettier": [
|
||||
"error",
|
||||
{
|
||||
"singleQuote": true,
|
||||
"endOfLine": "auto"
|
||||
}
|
||||
]
|
||||
},
|
||||
"overrides": [
|
||||
// Configuration for TypeScript files
|
||||
{
|
||||
"files": ["**/*.ts", "**/__tests__/*.test.ts"],
|
||||
"plugins": [
|
||||
"@typescript-eslint",
|
||||
"unused-imports",
|
||||
"simple-import-sort"
|
||||
],
|
||||
"extends": [
|
||||
"airbnb-typescript",
|
||||
"plugin:prettier/recommended"
|
||||
],
|
||||
"parserOptions": {
|
||||
"project": "./tsconfig.json"
|
||||
},
|
||||
"rules": {
|
||||
"prettier/prettier": [
|
||||
"error",
|
||||
{
|
||||
"singleQuote": true,
|
||||
"endOfLine": "auto"
|
||||
}
|
||||
],
|
||||
"@typescript-eslint/comma-dangle": "off", // Avoid conflict rule between Eslint and Prettier
|
||||
"@typescript-eslint/consistent-type-imports": "error", // Ensure `import type` is used when it's necessary
|
||||
"import/prefer-default-export": "off", // Named export is easier to refactor automatically
|
||||
"simple-import-sort/imports": "error", // Import configuration for `eslint-plugin-simple-import-sort`
|
||||
"simple-import-sort/exports": "error", // Export configuration for `eslint-plugin-simple-import-sort`
|
||||
"@typescript-eslint/no-unused-vars": "off",
|
||||
"react/jsx-filename-extension": "off", // Gives error
|
||||
"unused-imports/no-unused-imports": "error",
|
||||
"unused-imports/no-unused-vars": [
|
||||
"error",
|
||||
{ "argsIgnorePattern": "^_" }
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
-47
@@ -1,47 +0,0 @@
|
||||
name: Node.js Package
|
||||
|
||||
on:
|
||||
release:
|
||||
types: [created]
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/setup-node@v3
|
||||
with:
|
||||
node-version: 16
|
||||
- run: npm ci
|
||||
- run: npm test
|
||||
- run: npm run build
|
||||
- uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: dist
|
||||
path: dist
|
||||
- uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: types
|
||||
path: types
|
||||
|
||||
publish-npm:
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/setup-node@v3
|
||||
with:
|
||||
node-version: 16
|
||||
registry-url: https://registry.npmjs.org/
|
||||
- uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: dist
|
||||
path: dist
|
||||
- uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: types
|
||||
path: types
|
||||
- run: npm ci
|
||||
- run: npm publish
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{secrets.npm_token}}
|
||||
@@ -1,138 +0,0 @@
|
||||
# Logs
|
||||
logs
|
||||
*.log
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
lerna-debug.log*
|
||||
.pnpm-debug.log*
|
||||
|
||||
# Diagnostic reports (https://nodejs.org/api/report.html)
|
||||
report.[0-9]*.[0-9]*.[0-9]*.[0-9]*.json
|
||||
|
||||
# Runtime data
|
||||
pids
|
||||
*.pid
|
||||
*.seed
|
||||
*.pid.lock
|
||||
|
||||
# Directory for instrumented libs generated by jscoverage/JSCover
|
||||
lib-cov
|
||||
|
||||
# Coverage directory used by tools like istanbul
|
||||
coverage
|
||||
*.lcov
|
||||
|
||||
# nyc test coverage
|
||||
.nyc_output
|
||||
|
||||
# Grunt intermediate storage (https://gruntjs.com/creating-plugins#storing-task-files)
|
||||
.grunt
|
||||
|
||||
# Bower dependency directory (https://bower.io/)
|
||||
bower_components
|
||||
|
||||
# node-waf configuration
|
||||
.lock-wscript
|
||||
|
||||
# Compiled binary addons (https://nodejs.org/api/addons.html)
|
||||
build/Release
|
||||
|
||||
# Dependency directories
|
||||
node_modules/
|
||||
jspm_packages/
|
||||
|
||||
# Snowpack dependency directory (https://snowpack.dev/)
|
||||
web_modules/
|
||||
|
||||
# TypeScript cache
|
||||
*.tsbuildinfo
|
||||
|
||||
# Optional npm cache directory
|
||||
.npm
|
||||
|
||||
# Optional eslint cache
|
||||
.eslintcache
|
||||
|
||||
# Optional stylelint cache
|
||||
.stylelintcache
|
||||
|
||||
# Microbundle cache
|
||||
.rpt2_cache/
|
||||
.rts2_cache_cjs/
|
||||
.rts2_cache_es/
|
||||
.rts2_cache_umd/
|
||||
|
||||
# Optional REPL history
|
||||
.node_repl_history
|
||||
|
||||
# Output of 'npm pack'
|
||||
*.tgz
|
||||
|
||||
# Yarn Integrity file
|
||||
.yarn-integrity
|
||||
|
||||
# dotenv environment variable files
|
||||
.env
|
||||
.env.development.local
|
||||
.env.test.local
|
||||
.env.production.local
|
||||
.env.local
|
||||
|
||||
# parcel-bundler cache (https://parceljs.org/)
|
||||
.cache
|
||||
.parcel-cache
|
||||
|
||||
# Next.js build output
|
||||
.next
|
||||
out
|
||||
|
||||
# Nuxt.js build / generate output
|
||||
.nuxt
|
||||
dist
|
||||
|
||||
# Gatsby files
|
||||
.cache/
|
||||
# Comment in the public line in if your project uses Gatsby and not Next.js
|
||||
# https://nextjs.org/blog/next-9-1#public-directory-support
|
||||
# public
|
||||
|
||||
# vuepress build output
|
||||
.vuepress/dist
|
||||
|
||||
# vuepress v2.x temp and cache directory
|
||||
.temp
|
||||
.cache
|
||||
|
||||
# Docusaurus cache and generated files
|
||||
.docusaurus
|
||||
|
||||
# Serverless directories
|
||||
.serverless/
|
||||
|
||||
# FuseBox cache
|
||||
.fusebox/
|
||||
|
||||
# DynamoDB Local files
|
||||
.dynamodb/
|
||||
|
||||
# TernJS port file
|
||||
.tern-port
|
||||
|
||||
# Stores VSCode versions used for testing VSCode extensions
|
||||
.vscode-test
|
||||
|
||||
# yarn v2
|
||||
.yarn/cache
|
||||
.yarn/unplugged
|
||||
.yarn/build-state.yml
|
||||
.yarn/install-state.gz
|
||||
.pnp.*
|
||||
|
||||
.ideas.md
|
||||
.todos.md
|
||||
|
||||
# Custom
|
||||
dist
|
||||
types
|
||||
build
|
||||
@@ -1,4 +0,0 @@
|
||||
#!/bin/sh
|
||||
. "$(dirname "$0")/_/husky.sh"
|
||||
|
||||
npx --no -- commitlint --edit $1
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/bin/sh
|
||||
. "$(dirname "$0")/_/husky.sh"
|
||||
|
||||
# Disable concurent to run `check-types` after ESLint in lint-staged
|
||||
npx lint-staged --concurrent false
|
||||
@@ -1,8 +0,0 @@
|
||||
cff-version: 1.2.0
|
||||
message: "If you use this software, please cite it as below."
|
||||
authors:
|
||||
- family-names: "Singh"
|
||||
given-names: "Taranjeet"
|
||||
title: "Embedchain"
|
||||
date-released: 2023-06-25
|
||||
url: "https://github.com/embedchain/embedchainjs"
|
||||
@@ -1,201 +0,0 @@
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -1,254 +0,0 @@
|
||||
# embedchainjs
|
||||
|
||||
[](https://discord.gg/CUU9FPhRNt)
|
||||
[](https://twitter.com/embedchain)
|
||||
[](https://embedchain.substack.com/)
|
||||
|
||||
embedchain is a framework to easily create LLM powered bots over any dataset. embedchainjs is Javascript version of embedchain. If you want a python version, check out [embedchain-python](https://github.com/embedchain/embedchain)
|
||||
|
||||
# 🤝 Let's Talk Embedchain!
|
||||
|
||||
Schedule a [Feedback Session](https://cal.com/taranjeetio/ec) with Taranjeet, the founder, to discuss any issues, provide feedback, or explore improvements.
|
||||
|
||||
# How it works
|
||||
|
||||
It abstracts the entire process of loading dataset, chunking it, creating embeddings and then storing in vector database.
|
||||
|
||||
You can add a single or multiple dataset using `.add` and `.addLocal` function and then use `.query` function to find an answer from the added datasets.
|
||||
|
||||
If you want to create a Naval Ravikant bot which has 2 of his blog posts, as well as a question and answer pair you supply, all you need to do is add the links to the blog posts and the QnA pair and embedchain will create a bot for you.
|
||||
|
||||
```javascript
|
||||
const dotenv = require("dotenv");
|
||||
dotenv.config();
|
||||
const { App } = require("embedchain");
|
||||
|
||||
//Run the app commands inside an async function only
|
||||
async function testApp() {
|
||||
const navalChatBot = await App();
|
||||
|
||||
// Embed Online Resources
|
||||
await navalChatBot.add("web_page", "https://nav.al/feedback");
|
||||
await navalChatBot.add("web_page", "https://nav.al/agi");
|
||||
await navalChatBot.add(
|
||||
"pdf_file",
|
||||
"https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf"
|
||||
);
|
||||
|
||||
// Embed Local Resources
|
||||
await navalChatBot.addLocal("qna_pair", [
|
||||
"Who is Naval Ravikant?",
|
||||
"Naval Ravikant is an Indian-American entrepreneur and investor.",
|
||||
]);
|
||||
|
||||
const result = await navalChatBot.query(
|
||||
"What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?"
|
||||
);
|
||||
console.log(result);
|
||||
// answer: Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.
|
||||
}
|
||||
|
||||
testApp();
|
||||
```
|
||||
|
||||
# Getting Started
|
||||
|
||||
## Installation
|
||||
|
||||
- First make sure that you have the package installed. If not, then install it using `npm`
|
||||
|
||||
```bash
|
||||
npm install embedchain && npm install -S openai@^3.3.0
|
||||
```
|
||||
|
||||
- Currently, it is only compatible with openai 3.X, not the latest version 4.X. Please make sure to use the right version, otherwise you will see the `ChromaDB` error `TypeError: OpenAIApi.Configuration is not a constructor`
|
||||
|
||||
- Make sure that dotenv package is installed and your `OPENAI_API_KEY` in a file called `.env` in the root folder. You can install dotenv by
|
||||
|
||||
```js
|
||||
npm install dotenv
|
||||
```
|
||||
|
||||
- Download and install Docker on your device by visiting [this link](https://www.docker.com/). You will need this to run Chroma vector database on your machine.
|
||||
|
||||
- Run the following commands to setup Chroma container in Docker
|
||||
|
||||
```bash
|
||||
git clone https://github.com/chroma-core/chroma.git
|
||||
cd chroma
|
||||
docker-compose up -d --build
|
||||
```
|
||||
|
||||
- Once Chroma container has been set up, run it inside Docker
|
||||
|
||||
## Usage
|
||||
|
||||
- We use OpenAI's embedding model to create embeddings for chunks and ChatGPT API as LLM to get answer given the relevant docs. Make sure that you have an OpenAI account and an API key. If you have dont have an API key, you can create one by visiting [this link](https://platform.openai.com/account/api-keys).
|
||||
|
||||
- Once you have the API key, set it in an environment variable called `OPENAI_API_KEY`
|
||||
|
||||
```js
|
||||
// Set this inside your .env file
|
||||
OPENAI_API_KEY = "sk-xxxx";
|
||||
```
|
||||
|
||||
- Load the environment variables inside your .js file using the following commands
|
||||
|
||||
```js
|
||||
const dotenv = require("dotenv");
|
||||
dotenv.config();
|
||||
```
|
||||
|
||||
- Next import the `App` class from embedchain and use `.add` function to add any dataset.
|
||||
- Now your app is created. You can use `.query` function to get the answer for any query.
|
||||
|
||||
```js
|
||||
const dotenv = require("dotenv");
|
||||
dotenv.config();
|
||||
const { App } = require("embedchain");
|
||||
|
||||
async function testApp() {
|
||||
const navalChatBot = await App();
|
||||
|
||||
// Embed Online Resources
|
||||
await navalChatBot.add("web_page", "https://nav.al/feedback");
|
||||
await navalChatBot.add("web_page", "https://nav.al/agi");
|
||||
await navalChatBot.add(
|
||||
"pdf_file",
|
||||
"https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf"
|
||||
);
|
||||
|
||||
// Embed Local Resources
|
||||
await navalChatBot.addLocal("qna_pair", [
|
||||
"Who is Naval Ravikant?",
|
||||
"Naval Ravikant is an Indian-American entrepreneur and investor.",
|
||||
]);
|
||||
|
||||
const result = await navalChatBot.query(
|
||||
"What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?"
|
||||
);
|
||||
console.log(result);
|
||||
// answer: Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.
|
||||
}
|
||||
|
||||
testApp();
|
||||
```
|
||||
|
||||
- If there is any other app instance in your script or app, you can change the import as
|
||||
|
||||
```javascript
|
||||
const { App: EmbedChainApp } = require("embedchain");
|
||||
|
||||
// or
|
||||
|
||||
const { App: ECApp } = require("embedchain");
|
||||
```
|
||||
|
||||
## Format supported
|
||||
|
||||
We support the following formats:
|
||||
|
||||
### PDF File
|
||||
|
||||
To add any pdf file, use the data_type as `pdf_file`. Eg:
|
||||
|
||||
```javascript
|
||||
await app.add("pdf_file", "a_valid_url_where_pdf_file_can_be_accessed");
|
||||
```
|
||||
|
||||
### Web Page
|
||||
|
||||
To add any web page, use the data_type as `web_page`. Eg:
|
||||
|
||||
```javascript
|
||||
await app.add("web_page", "a_valid_web_page_url");
|
||||
```
|
||||
|
||||
### QnA Pair
|
||||
|
||||
To supply your own QnA pair, use the data_type as `qna_pair` and enter a tuple. Eg:
|
||||
|
||||
```javascript
|
||||
await app.addLocal("qna_pair", ["Question", "Answer"]);
|
||||
```
|
||||
|
||||
### More Formats coming soon
|
||||
|
||||
- If you want to add any other format, please create an [issue](https://github.com/embedchain/embedchainjs/issues) and we will add it to the list of supported formats.
|
||||
|
||||
## Testing
|
||||
|
||||
Before you consume valuable tokens, you should make sure that the embedding you have done works and that it's receiving the correct document from the database.
|
||||
|
||||
For this you can use the `dryRun` method.
|
||||
|
||||
Following the example above, add this to your script:
|
||||
|
||||
```js
|
||||
let result = await naval_chat_bot.dryRun("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?");console.log(result);
|
||||
|
||||
'''
|
||||
Use the following pieces of context to answer the query at the end. If you don't know the answer, just say that you don't know, don't try to make up an answer.
|
||||
terms of the unseen. And I think that’s critical. That is what humans do uniquely that no other creature, no other computer, no other intelligence—biological or artificial—that we have ever encountered does. And not only do we do it uniquely, but if we were to meet an alien species that also had the power to generate these good explanations, there is no explanation that they could generate that we could not understand. We are maximally capable of understanding. There is no concept out there that is possible in this physical reality that a human being, given sufficient time and resources and
|
||||
Query: What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?
|
||||
Helpful Answer:
|
||||
'''
|
||||
```
|
||||
|
||||
_The embedding is confirmed to work as expected. It returns the right document, even if the question is asked slightly different. No prompt tokens have been consumed._
|
||||
|
||||
**The dry run will still consume tokens to embed your query, but it is only ~1/15 of the prompt.**
|
||||
|
||||
# How does it work?
|
||||
|
||||
Creating a chat bot over any dataset needs the following steps to happen
|
||||
|
||||
- load the data
|
||||
- create meaningful chunks
|
||||
- create embeddings for each chunk
|
||||
- store the chunks in vector database
|
||||
|
||||
Whenever a user asks any query, following process happens to find the answer for the query
|
||||
|
||||
- create the embedding for query
|
||||
- find similar documents for this query from vector database
|
||||
- pass similar documents as context to LLM to get the final answer.
|
||||
|
||||
The process of loading the dataset and then querying involves multiple steps and each steps has nuances of it is own.
|
||||
|
||||
- How should I chunk the data? What is a meaningful chunk size?
|
||||
- How should I create embeddings for each chunk? Which embedding model should I use?
|
||||
- How should I store the chunks in vector database? Which vector database should I use?
|
||||
- Should I store meta data along with the embeddings?
|
||||
- How should I find similar documents for a query? Which ranking model should I use?
|
||||
|
||||
These questions may be trivial for some but for a lot of us, it needs research, experimentation and time to find out the accurate answers.
|
||||
|
||||
embedchain is a framework which takes care of all these nuances and provides a simple interface to create bots over any dataset.
|
||||
|
||||
In the first release, we are making it easier for anyone to get a chatbot over any dataset up and running in less than a minute. All you need to do is create an app instance, add the data sets using `.add` function and then use `.query` function to get the relevant answer.
|
||||
|
||||
# Team
|
||||
|
||||
## Author
|
||||
|
||||
- Taranjeet Singh ([@taranjeetio](https://twitter.com/taranjeetio))
|
||||
|
||||
## Maintainer
|
||||
|
||||
- [cachho](https://github.com/cachho)
|
||||
- [sahilyadav902](https://github.com/sahilyadav902)
|
||||
|
||||
## Citation
|
||||
|
||||
If you utilize this repository, please consider citing it with:
|
||||
```
|
||||
@misc{embedchain,
|
||||
author = {Taranjeet Singh},
|
||||
title = {Embechain: Framework to easily create LLM powered bots over any dataset},
|
||||
year = {2023},
|
||||
publisher = {GitHub},
|
||||
journal = {GitHub repository},
|
||||
howpublished = {\url{https://github.com/embedchain/embedchainjs}},
|
||||
}
|
||||
```
|
||||
@@ -1 +0,0 @@
|
||||
module.exports = { extends: ['@commitlint/config-conventional'] };
|
||||
@@ -1,66 +0,0 @@
|
||||
import { EmbedChainApp } from '../embedchain';
|
||||
|
||||
const mockAdd = jest.fn();
|
||||
const mockAddLocal = jest.fn();
|
||||
const mockQuery = jest.fn();
|
||||
|
||||
jest.mock('../embedchain', () => {
|
||||
return {
|
||||
EmbedChainApp: jest.fn().mockImplementation(() => {
|
||||
return {
|
||||
add: mockAdd,
|
||||
addLocal: mockAddLocal,
|
||||
query: mockQuery,
|
||||
};
|
||||
}),
|
||||
};
|
||||
});
|
||||
|
||||
describe('Test App', () => {
|
||||
beforeEach(() => {
|
||||
jest.clearAllMocks();
|
||||
});
|
||||
|
||||
it('tests the App', async () => {
|
||||
mockQuery.mockResolvedValue(
|
||||
'Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.'
|
||||
);
|
||||
|
||||
const navalChatBot = await new EmbedChainApp(undefined, false);
|
||||
|
||||
// Embed Online Resources
|
||||
await navalChatBot.add('web_page', 'https://nav.al/feedback');
|
||||
await navalChatBot.add('web_page', 'https://nav.al/agi');
|
||||
await navalChatBot.add(
|
||||
'pdf_file',
|
||||
'https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf'
|
||||
);
|
||||
|
||||
// Embed Local Resources
|
||||
await navalChatBot.addLocal('qna_pair', [
|
||||
'Who is Naval Ravikant?',
|
||||
'Naval Ravikant is an Indian-American entrepreneur and investor.',
|
||||
]);
|
||||
|
||||
const result = await navalChatBot.query(
|
||||
'What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?'
|
||||
);
|
||||
|
||||
expect(mockAdd).toHaveBeenCalledWith('web_page', 'https://nav.al/feedback');
|
||||
expect(mockAdd).toHaveBeenCalledWith('web_page', 'https://nav.al/agi');
|
||||
expect(mockAdd).toHaveBeenCalledWith(
|
||||
'pdf_file',
|
||||
'https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf'
|
||||
);
|
||||
expect(mockAddLocal).toHaveBeenCalledWith('qna_pair', [
|
||||
'Who is Naval Ravikant?',
|
||||
'Naval Ravikant is an Indian-American entrepreneur and investor.',
|
||||
]);
|
||||
expect(mockQuery).toHaveBeenCalledWith(
|
||||
'What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?'
|
||||
);
|
||||
expect(result).toBe(
|
||||
'Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.'
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -1,44 +0,0 @@
|
||||
import { createHash } from 'crypto';
|
||||
import type { RecursiveCharacterTextSplitter } from 'langchain/text_splitter';
|
||||
|
||||
import type { BaseLoader } from '../loaders';
|
||||
import type { Input, LoaderResult } from '../models';
|
||||
import type { ChunkResult } from '../models/ChunkResult';
|
||||
|
||||
class BaseChunker {
|
||||
textSplitter: RecursiveCharacterTextSplitter;
|
||||
|
||||
constructor(textSplitter: RecursiveCharacterTextSplitter) {
|
||||
this.textSplitter = textSplitter;
|
||||
}
|
||||
|
||||
async createChunks(loader: BaseLoader, url: Input): Promise<ChunkResult> {
|
||||
const documents: ChunkResult['documents'] = [];
|
||||
const ids: ChunkResult['ids'] = [];
|
||||
const datas: LoaderResult = await loader.loadData(url);
|
||||
const metadatas: ChunkResult['metadatas'] = [];
|
||||
|
||||
const dataPromises = datas.map(async (data) => {
|
||||
const { content, metaData } = data;
|
||||
const chunks: string[] = await this.textSplitter.splitText(content);
|
||||
chunks.forEach((chunk) => {
|
||||
const chunkId = createHash('sha256')
|
||||
.update(chunk + metaData.url)
|
||||
.digest('hex');
|
||||
ids.push(chunkId);
|
||||
documents.push(chunk);
|
||||
metadatas.push(metaData);
|
||||
});
|
||||
});
|
||||
|
||||
await Promise.all(dataPromises);
|
||||
|
||||
return {
|
||||
documents,
|
||||
ids,
|
||||
metadatas,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export { BaseChunker };
|
||||
@@ -1,26 +0,0 @@
|
||||
import { RecursiveCharacterTextSplitter } from 'langchain/text_splitter';
|
||||
|
||||
import { BaseChunker } from './BaseChunker';
|
||||
|
||||
interface TextSplitterChunkParams {
|
||||
chunkSize: number;
|
||||
chunkOverlap: number;
|
||||
keepSeparator: boolean;
|
||||
}
|
||||
|
||||
const TEXT_SPLITTER_CHUNK_PARAMS: TextSplitterChunkParams = {
|
||||
chunkSize: 1000,
|
||||
chunkOverlap: 0,
|
||||
keepSeparator: false,
|
||||
};
|
||||
|
||||
class PdfFileChunker extends BaseChunker {
|
||||
constructor() {
|
||||
const textSplitter = new RecursiveCharacterTextSplitter(
|
||||
TEXT_SPLITTER_CHUNK_PARAMS
|
||||
);
|
||||
super(textSplitter);
|
||||
}
|
||||
}
|
||||
|
||||
export { PdfFileChunker };
|
||||
@@ -1,26 +0,0 @@
|
||||
import { RecursiveCharacterTextSplitter } from 'langchain/text_splitter';
|
||||
|
||||
import { BaseChunker } from './BaseChunker';
|
||||
|
||||
interface TextSplitterChunkParams {
|
||||
chunkSize: number;
|
||||
chunkOverlap: number;
|
||||
keepSeparator: boolean;
|
||||
}
|
||||
|
||||
const TEXT_SPLITTER_CHUNK_PARAMS: TextSplitterChunkParams = {
|
||||
chunkSize: 300,
|
||||
chunkOverlap: 0,
|
||||
keepSeparator: false,
|
||||
};
|
||||
|
||||
class QnaPairChunker extends BaseChunker {
|
||||
constructor() {
|
||||
const textSplitter = new RecursiveCharacterTextSplitter(
|
||||
TEXT_SPLITTER_CHUNK_PARAMS
|
||||
);
|
||||
super(textSplitter);
|
||||
}
|
||||
}
|
||||
|
||||
export { QnaPairChunker };
|
||||
@@ -1,26 +0,0 @@
|
||||
import { RecursiveCharacterTextSplitter } from 'langchain/text_splitter';
|
||||
|
||||
import { BaseChunker } from './BaseChunker';
|
||||
|
||||
interface TextSplitterChunkParams {
|
||||
chunkSize: number;
|
||||
chunkOverlap: number;
|
||||
keepSeparator: boolean;
|
||||
}
|
||||
|
||||
const TEXT_SPLITTER_CHUNK_PARAMS: TextSplitterChunkParams = {
|
||||
chunkSize: 500,
|
||||
chunkOverlap: 0,
|
||||
keepSeparator: false,
|
||||
};
|
||||
|
||||
class WebPageChunker extends BaseChunker {
|
||||
constructor() {
|
||||
const textSplitter = new RecursiveCharacterTextSplitter(
|
||||
TEXT_SPLITTER_CHUNK_PARAMS
|
||||
);
|
||||
super(textSplitter);
|
||||
}
|
||||
}
|
||||
|
||||
export { WebPageChunker };
|
||||
@@ -1,6 +0,0 @@
|
||||
import { BaseChunker } from './BaseChunker';
|
||||
import { PdfFileChunker } from './PdfFile';
|
||||
import { QnaPairChunker } from './QnaPair';
|
||||
import { WebPageChunker } from './WebPage';
|
||||
|
||||
export { BaseChunker, PdfFileChunker, QnaPairChunker, WebPageChunker };
|
||||
@@ -1,317 +0,0 @@
|
||||
/* eslint-disable max-classes-per-file */
|
||||
import type { Collection } from 'chromadb';
|
||||
import type { QueryResponse } from 'chromadb/dist/main/types';
|
||||
import * as fs from 'fs';
|
||||
import { Document } from 'langchain/document';
|
||||
import OpenAI from 'openai';
|
||||
import * as path from 'path';
|
||||
import { v4 as uuidv4 } from 'uuid';
|
||||
|
||||
import type { BaseChunker } from './chunkers';
|
||||
import { PdfFileChunker, QnaPairChunker, WebPageChunker } from './chunkers';
|
||||
import type { BaseLoader } from './loaders';
|
||||
import { LocalQnaPairLoader, PdfFileLoader, WebPageLoader } from './loaders';
|
||||
import type {
|
||||
DataDict,
|
||||
DataType,
|
||||
FormattedResult,
|
||||
Input,
|
||||
LocalInput,
|
||||
Metadata,
|
||||
Method,
|
||||
RemoteInput,
|
||||
} from './models';
|
||||
import { ChromaDB } from './vectordb';
|
||||
import type { BaseVectorDB } from './vectordb/BaseVectorDb';
|
||||
|
||||
const openai = new OpenAI({
|
||||
apiKey: process.env.OPENAI_API_KEY,
|
||||
});
|
||||
|
||||
class EmbedChain {
|
||||
dbClient: any;
|
||||
|
||||
// TODO: Definitely assign
|
||||
collection!: Collection;
|
||||
|
||||
userAsks: [DataType, Input][] = [];
|
||||
|
||||
initApp: Promise<void>;
|
||||
|
||||
collectMetrics: boolean;
|
||||
|
||||
sId: string; // sessionId
|
||||
|
||||
constructor(db?: BaseVectorDB, collectMetrics: boolean = true) {
|
||||
if (!db) {
|
||||
this.initApp = this.setupChroma();
|
||||
} else {
|
||||
this.initApp = this.setupOther(db);
|
||||
}
|
||||
|
||||
this.collectMetrics = collectMetrics;
|
||||
|
||||
// Send anonymous telemetry
|
||||
this.sId = uuidv4();
|
||||
this.sendTelemetryEvent('init');
|
||||
}
|
||||
|
||||
async setupChroma(): Promise<void> {
|
||||
const db = new ChromaDB();
|
||||
await db.initDb;
|
||||
this.dbClient = db.client;
|
||||
if (db.collection) {
|
||||
this.collection = db.collection;
|
||||
} else {
|
||||
// TODO: Add proper error handling
|
||||
console.error('No collection');
|
||||
}
|
||||
}
|
||||
|
||||
async setupOther(db: BaseVectorDB): Promise<void> {
|
||||
await db.initDb;
|
||||
// TODO: Figure out how we can initialize an unknown database.
|
||||
// this.dbClient = db.client;
|
||||
// this.collection = db.collection;
|
||||
this.userAsks = [];
|
||||
}
|
||||
|
||||
static getLoader(dataType: DataType) {
|
||||
const loaders: { [t in DataType]: BaseLoader } = {
|
||||
pdf_file: new PdfFileLoader(),
|
||||
web_page: new WebPageLoader(),
|
||||
qna_pair: new LocalQnaPairLoader(),
|
||||
};
|
||||
return loaders[dataType];
|
||||
}
|
||||
|
||||
static getChunker(dataType: DataType) {
|
||||
const chunkers: { [t in DataType]: BaseChunker } = {
|
||||
pdf_file: new PdfFileChunker(),
|
||||
web_page: new WebPageChunker(),
|
||||
qna_pair: new QnaPairChunker(),
|
||||
};
|
||||
return chunkers[dataType];
|
||||
}
|
||||
|
||||
public async add(dataType: DataType, url: RemoteInput) {
|
||||
const loader = EmbedChain.getLoader(dataType);
|
||||
const chunker = EmbedChain.getChunker(dataType);
|
||||
this.userAsks.push([dataType, url]);
|
||||
const { documents, countNewChunks } = await this.loadAndEmbed(
|
||||
loader,
|
||||
chunker,
|
||||
url
|
||||
);
|
||||
|
||||
if (this.collectMetrics) {
|
||||
const wordCount = documents.reduce(
|
||||
(sum, document) => sum + document.split(' ').length,
|
||||
0
|
||||
);
|
||||
|
||||
this.sendTelemetryEvent('add', {
|
||||
data_type: dataType,
|
||||
word_count: wordCount,
|
||||
chunks_count: countNewChunks,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
public async addLocal(dataType: DataType, content: LocalInput) {
|
||||
const loader = EmbedChain.getLoader(dataType);
|
||||
const chunker = EmbedChain.getChunker(dataType);
|
||||
this.userAsks.push([dataType, content]);
|
||||
const { documents, countNewChunks } = await this.loadAndEmbed(
|
||||
loader,
|
||||
chunker,
|
||||
content
|
||||
);
|
||||
|
||||
if (this.collectMetrics) {
|
||||
const wordCount = documents.reduce(
|
||||
(sum, document) => sum + document.split(' ').length,
|
||||
0
|
||||
);
|
||||
|
||||
this.sendTelemetryEvent('add_local', {
|
||||
data_type: dataType,
|
||||
word_count: wordCount,
|
||||
chunks_count: countNewChunks,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
protected async loadAndEmbed(
|
||||
loader: any,
|
||||
chunker: BaseChunker,
|
||||
src: Input
|
||||
): Promise<{
|
||||
documents: string[];
|
||||
metadatas: Metadata[];
|
||||
ids: string[];
|
||||
countNewChunks: number;
|
||||
}> {
|
||||
const embeddingsData = await chunker.createChunks(loader, src);
|
||||
let { documents, ids, metadatas } = embeddingsData;
|
||||
|
||||
const existingDocs = await this.collection.get({ ids });
|
||||
const existingIds = new Set(existingDocs.ids);
|
||||
|
||||
if (existingIds.size > 0) {
|
||||
const dataDict: DataDict = {};
|
||||
for (let i = 0; i < ids.length; i += 1) {
|
||||
const id = ids[i];
|
||||
if (!existingIds.has(id)) {
|
||||
dataDict[id] = { doc: documents[i], meta: metadatas[i] };
|
||||
}
|
||||
}
|
||||
|
||||
if (Object.keys(dataDict).length === 0) {
|
||||
console.log(`All data from ${src} already exists in the database.`);
|
||||
return { documents: [], metadatas: [], ids: [], countNewChunks: 0 };
|
||||
}
|
||||
ids = Object.keys(dataDict);
|
||||
const dataValues = Object.values(dataDict);
|
||||
documents = dataValues.map(({ doc }) => doc);
|
||||
metadatas = dataValues.map(({ meta }) => meta);
|
||||
}
|
||||
|
||||
const countBeforeAddition = await this.count();
|
||||
await this.collection.add({ documents, metadatas, ids });
|
||||
const countNewChunks = (await this.count()) - countBeforeAddition;
|
||||
console.log(
|
||||
`Successfully saved ${src}. New chunks count: ${countNewChunks}`
|
||||
);
|
||||
return { documents, metadatas, ids, countNewChunks };
|
||||
}
|
||||
|
||||
static async formatResult(
|
||||
results: QueryResponse
|
||||
): Promise<FormattedResult[]> {
|
||||
return results.documents[0].map((document: any, index: number) => {
|
||||
const metadata = results.metadatas[0][index] || {};
|
||||
// TODO: Add proper error handling
|
||||
const distance = results.distances ? results.distances[0][index] : null;
|
||||
return [new Document({ pageContent: document, metadata }), distance];
|
||||
});
|
||||
}
|
||||
|
||||
static async getOpenAiAnswer(prompt: string) {
|
||||
const messages: OpenAI.Chat.CreateChatCompletionRequestMessage[] = [
|
||||
{ role: 'user', content: prompt },
|
||||
];
|
||||
const response = await openai.chat.completions.create({
|
||||
model: 'gpt-3.5-turbo',
|
||||
messages,
|
||||
temperature: 0,
|
||||
max_tokens: 1000,
|
||||
top_p: 1,
|
||||
});
|
||||
return (
|
||||
response.choices[0].message?.content ?? 'Response could not be processed.'
|
||||
);
|
||||
}
|
||||
|
||||
protected async retrieveFromDatabase(inputQuery: string) {
|
||||
const result = await this.collection.query({
|
||||
nResults: 1,
|
||||
queryTexts: [inputQuery],
|
||||
});
|
||||
const resultFormatted = await EmbedChain.formatResult(result);
|
||||
const content = resultFormatted[0][0].pageContent;
|
||||
return content;
|
||||
}
|
||||
|
||||
static generatePrompt(inputQuery: string, context: any) {
|
||||
const prompt = `Use the following pieces of context to answer the query at the end. If you don't know the answer, just say that you don't know, don't try to make up an answer.\n${context}\nQuery: ${inputQuery}\nHelpful Answer:`;
|
||||
return prompt;
|
||||
}
|
||||
|
||||
static async getAnswerFromLlm(prompt: string) {
|
||||
const answer = await EmbedChain.getOpenAiAnswer(prompt);
|
||||
return answer;
|
||||
}
|
||||
|
||||
public async query(inputQuery: string) {
|
||||
const context = await this.retrieveFromDatabase(inputQuery);
|
||||
const prompt = EmbedChain.generatePrompt(inputQuery, context);
|
||||
const answer = await EmbedChain.getAnswerFromLlm(prompt);
|
||||
this.sendTelemetryEvent('query');
|
||||
return answer;
|
||||
}
|
||||
|
||||
public async dryRun(input_query: string) {
|
||||
const context = await this.retrieveFromDatabase(input_query);
|
||||
const prompt = EmbedChain.generatePrompt(input_query, context);
|
||||
return prompt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Count the number of embeddings.
|
||||
* @returns {Promise<number>}: The number of embeddings.
|
||||
*/
|
||||
public count(): Promise<number> {
|
||||
return this.collection.count();
|
||||
}
|
||||
|
||||
protected async sendTelemetryEvent(method: Method, extraMetadata?: object) {
|
||||
if (!this.collectMetrics) {
|
||||
return;
|
||||
}
|
||||
const url = 'https://api.embedchain.ai/api/v1/telemetry/';
|
||||
|
||||
// Read package version from filesystem (because it's not in the ts root dir)
|
||||
const packageJsonPath = path.join(__dirname, '..', 'package.json');
|
||||
const packageJson = JSON.parse(fs.readFileSync(packageJsonPath, 'utf8'));
|
||||
|
||||
const metadata = {
|
||||
s_id: this.sId,
|
||||
version: packageJson.version,
|
||||
method,
|
||||
language: 'js',
|
||||
...extraMetadata,
|
||||
};
|
||||
|
||||
const maxRetries = 3;
|
||||
|
||||
// Retry the fetch
|
||||
for (let i = 0; i < maxRetries; i += 1) {
|
||||
try {
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
const response = await fetch(url, {
|
||||
method: 'POST',
|
||||
body: JSON.stringify({ metadata }),
|
||||
});
|
||||
|
||||
if (response.ok) {
|
||||
// Break out of the loop if the request was successful
|
||||
break;
|
||||
} else {
|
||||
// Log the unsuccessful response (optional)
|
||||
console.error(
|
||||
`Telemetry: Attempt ${i + 1} failed with status:`,
|
||||
response.status
|
||||
);
|
||||
}
|
||||
} catch (error) {
|
||||
// Log the error (optional)
|
||||
console.error(`Telemetry: Attempt ${i + 1} failed with error:`, error);
|
||||
}
|
||||
|
||||
// If this was the last attempt, throw an error or handle the failure
|
||||
if (i === maxRetries - 1) {
|
||||
console.error('Telemetry: Max retries reached');
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
class EmbedChainApp extends EmbedChain {
|
||||
// The EmbedChain app.
|
||||
// Has two functions: add and query.
|
||||
// adds(dataType, url): adds the data from the given URL to the vector db.
|
||||
// query(query): finds answer to the given query using vector database and LLM.
|
||||
}
|
||||
|
||||
export { EmbedChainApp };
|
||||
@@ -1,7 +0,0 @@
|
||||
import { EmbedChainApp } from './embedchain';
|
||||
|
||||
export const App = async () => {
|
||||
const app = new EmbedChainApp();
|
||||
await app.initApp;
|
||||
return app;
|
||||
};
|
||||
@@ -1,5 +0,0 @@
|
||||
import type { Input, LoaderResult } from '../models';
|
||||
|
||||
export abstract class BaseLoader {
|
||||
abstract loadData(src: Input): Promise<LoaderResult>;
|
||||
}
|
||||
@@ -1,21 +0,0 @@
|
||||
import type { LoaderResult, QnaPair } from '../models';
|
||||
import { BaseLoader } from './BaseLoader';
|
||||
|
||||
class LocalQnaPairLoader extends BaseLoader {
|
||||
// eslint-disable-next-line class-methods-use-this
|
||||
async loadData(content: QnaPair): Promise<LoaderResult> {
|
||||
const [question, answer] = content;
|
||||
const contentText = `Q: ${question}\nA: ${answer}`;
|
||||
const metaData = {
|
||||
url: 'local',
|
||||
};
|
||||
return [
|
||||
{
|
||||
content: contentText,
|
||||
metaData,
|
||||
},
|
||||
];
|
||||
}
|
||||
}
|
||||
|
||||
export { LocalQnaPairLoader };
|
||||
@@ -1,58 +0,0 @@
|
||||
import type { TextContent } from 'pdfjs-dist/types/src/display/api';
|
||||
|
||||
import type { LoaderResult, Metadata } from '../models';
|
||||
import { cleanString } from '../utils';
|
||||
import { BaseLoader } from './BaseLoader';
|
||||
|
||||
const pdfjsLib = require('pdfjs-dist');
|
||||
|
||||
interface Page {
|
||||
page_content: string;
|
||||
}
|
||||
|
||||
class PdfFileLoader extends BaseLoader {
|
||||
static async getPagesFromPdf(url: string): Promise<Page[]> {
|
||||
const loadingTask = pdfjsLib.getDocument(url);
|
||||
const pdf = await loadingTask.promise;
|
||||
const { numPages } = pdf;
|
||||
|
||||
const promises = Array.from({ length: numPages }, async (_, i) => {
|
||||
const page = await pdf.getPage(i + 1);
|
||||
const pageText: TextContent = await page.getTextContent();
|
||||
const pageContent: string = pageText.items
|
||||
.map((item) => ('str' in item ? item.str : ''))
|
||||
.join(' ');
|
||||
|
||||
return {
|
||||
page_content: pageContent,
|
||||
};
|
||||
});
|
||||
|
||||
return Promise.all(promises);
|
||||
}
|
||||
|
||||
// eslint-disable-next-line class-methods-use-this
|
||||
async loadData(url: string): Promise<LoaderResult> {
|
||||
const pages: Page[] = await PdfFileLoader.getPagesFromPdf(url);
|
||||
const output: LoaderResult = [];
|
||||
|
||||
if (!pages.length) {
|
||||
throw new Error('No data found');
|
||||
}
|
||||
|
||||
pages.forEach((page) => {
|
||||
let content: string = page.page_content;
|
||||
content = cleanString(content);
|
||||
const metaData: Metadata = {
|
||||
url,
|
||||
};
|
||||
output.push({
|
||||
content,
|
||||
metaData,
|
||||
});
|
||||
});
|
||||
return output;
|
||||
}
|
||||
}
|
||||
|
||||
export { PdfFileLoader };
|
||||
@@ -1,51 +0,0 @@
|
||||
import axios from 'axios';
|
||||
import { JSDOM } from 'jsdom';
|
||||
|
||||
import { cleanString } from '../utils';
|
||||
import { BaseLoader } from './BaseLoader';
|
||||
|
||||
class WebPageLoader extends BaseLoader {
|
||||
// eslint-disable-next-line class-methods-use-this
|
||||
async loadData(url: string) {
|
||||
const response = await axios.get(url);
|
||||
const html = response.data;
|
||||
const dom = new JSDOM(html);
|
||||
const { document } = dom.window;
|
||||
const unwantedTags = [
|
||||
'nav',
|
||||
'aside',
|
||||
'form',
|
||||
'header',
|
||||
'noscript',
|
||||
'svg',
|
||||
'canvas',
|
||||
'footer',
|
||||
'script',
|
||||
'style',
|
||||
];
|
||||
unwantedTags.forEach((tagName) => {
|
||||
const elements = document.getElementsByTagName(tagName);
|
||||
Array.from(elements).forEach((element) => {
|
||||
// eslint-disable-next-line no-param-reassign
|
||||
(element as HTMLElement).textContent = ' ';
|
||||
});
|
||||
});
|
||||
|
||||
const output = [];
|
||||
let content = document.body.textContent;
|
||||
if (!content) {
|
||||
throw new Error('Web page content is empty.');
|
||||
}
|
||||
content = cleanString(content);
|
||||
const metaData = {
|
||||
url,
|
||||
};
|
||||
output.push({
|
||||
content,
|
||||
metaData,
|
||||
});
|
||||
return output;
|
||||
}
|
||||
}
|
||||
|
||||
export { WebPageLoader };
|
||||
@@ -1,6 +0,0 @@
|
||||
import { BaseLoader } from './BaseLoader';
|
||||
import { LocalQnaPairLoader } from './LocalQnaPair';
|
||||
import { PdfFileLoader } from './PdfFile';
|
||||
import { WebPageLoader } from './WebPage';
|
||||
|
||||
export { BaseLoader, LocalQnaPairLoader, PdfFileLoader, WebPageLoader };
|
||||
@@ -1,7 +0,0 @@
|
||||
import type { Metadata } from './Metadata';
|
||||
|
||||
export type ChunkResult = {
|
||||
documents: string[];
|
||||
ids: string[];
|
||||
metadatas: Metadata[];
|
||||
};
|
||||
@@ -1,10 +0,0 @@
|
||||
import type { ChunkResult } from './ChunkResult';
|
||||
|
||||
type Data = {
|
||||
doc: ChunkResult['documents'][0];
|
||||
meta: ChunkResult['metadatas'][0];
|
||||
};
|
||||
|
||||
export type DataDict = {
|
||||
[id: string]: Data;
|
||||
};
|
||||
@@ -1 +0,0 @@
|
||||
export type DataType = 'pdf_file' | 'web_page' | 'qna_pair';
|
||||
@@ -1,3 +0,0 @@
|
||||
import type { Document } from 'langchain/document';
|
||||
|
||||
export type FormattedResult = [Document, number | null];
|
||||
@@ -1,7 +0,0 @@
|
||||
import type { QnaPair } from './QnAPair';
|
||||
|
||||
export type RemoteInput = string;
|
||||
|
||||
export type LocalInput = QnaPair;
|
||||
|
||||
export type Input = RemoteInput | LocalInput;
|
||||
@@ -1,3 +0,0 @@
|
||||
import type { Metadata } from './Metadata';
|
||||
|
||||
export type LoaderResult = { content: any; metaData: Metadata }[];
|
||||
@@ -1,3 +0,0 @@
|
||||
export type Metadata = {
|
||||
url: string;
|
||||
};
|
||||
@@ -1 +0,0 @@
|
||||
export type Method = 'init' | 'query' | 'add' | 'add_local';
|
||||
@@ -1,4 +0,0 @@
|
||||
type Question = string;
|
||||
type Answer = string;
|
||||
|
||||
export type QnaPair = [Question, Answer];
|
||||
@@ -1,21 +0,0 @@
|
||||
import { DataDict } from './DataDict';
|
||||
import { DataType } from './DataType';
|
||||
import { FormattedResult } from './FormattedResult';
|
||||
import { Input, LocalInput, RemoteInput } from './Input';
|
||||
import { LoaderResult } from './LoaderResult';
|
||||
import { Metadata } from './Metadata';
|
||||
import { Method } from './Method';
|
||||
import { QnaPair } from './QnAPair';
|
||||
|
||||
export {
|
||||
DataDict,
|
||||
DataType,
|
||||
FormattedResult,
|
||||
Input,
|
||||
LoaderResult,
|
||||
LocalInput,
|
||||
Metadata,
|
||||
Method,
|
||||
QnaPair,
|
||||
RemoteInput,
|
||||
};
|
||||
@@ -1,26 +0,0 @@
|
||||
/**
|
||||
* This function takes in a string and performs a series of text cleaning operations.
|
||||
* @param {str} text: The text to be cleaned. This is expected to be a string.
|
||||
* @returns {str}: The cleaned text after all the cleaning operations have been performed.
|
||||
*/
|
||||
export function cleanString(text: string): string {
|
||||
// Replacement of newline characters:
|
||||
let cleanedText = text.replace(/\n/g, ' ');
|
||||
|
||||
// Stripping and reducing multiple spaces to single:
|
||||
cleanedText = cleanedText.trim().replace(/\s+/g, ' ');
|
||||
|
||||
// Removing backslashes:
|
||||
cleanedText = cleanedText.replace(/\\/g, '');
|
||||
|
||||
// Replacing hash characters:
|
||||
cleanedText = cleanedText.replace(/#/g, ' ');
|
||||
|
||||
// Eliminating consecutive non-alphanumeric characters:
|
||||
// This regex identifies consecutive non-alphanumeric characters (i.e., not a word character [a-zA-Z0-9_] and not a whitespace) in the string
|
||||
// and replaces each group of such characters with a single occurrence of that character.
|
||||
// For example, "!!! hello !!!" would become "! hello !".
|
||||
cleanedText = cleanedText.replace(/([^\w\s])\1*/g, '$1');
|
||||
|
||||
return cleanedText;
|
||||
}
|
||||
@@ -1,14 +0,0 @@
|
||||
class BaseVectorDB {
|
||||
initDb: Promise<void>;
|
||||
|
||||
constructor() {
|
||||
this.initDb = this.getClientAndCollection();
|
||||
}
|
||||
|
||||
// eslint-disable-next-line class-methods-use-this
|
||||
protected async getClientAndCollection(): Promise<void> {
|
||||
throw new Error('getClientAndCollection() method is not implemented');
|
||||
}
|
||||
}
|
||||
|
||||
export { BaseVectorDB };
|
||||
@@ -1,38 +0,0 @@
|
||||
import type { Collection } from 'chromadb';
|
||||
import { ChromaClient, OpenAIEmbeddingFunction } from 'chromadb';
|
||||
|
||||
import { BaseVectorDB } from './BaseVectorDb';
|
||||
|
||||
const embedder = new OpenAIEmbeddingFunction({
|
||||
openai_api_key: process.env.OPENAI_API_KEY ?? '',
|
||||
});
|
||||
|
||||
class ChromaDB extends BaseVectorDB {
|
||||
client: ChromaClient | undefined;
|
||||
|
||||
collection: Collection | null = null;
|
||||
|
||||
// eslint-disable-next-line @typescript-eslint/no-useless-constructor
|
||||
constructor() {
|
||||
super();
|
||||
}
|
||||
|
||||
protected async getClientAndCollection(): Promise<void> {
|
||||
this.client = new ChromaClient({ path: 'http://localhost:8000' });
|
||||
try {
|
||||
this.collection = await this.client.getCollection({
|
||||
name: 'embedchain_store',
|
||||
embeddingFunction: embedder,
|
||||
});
|
||||
} catch (err) {
|
||||
if (!this.collection) {
|
||||
this.collection = await this.client.createCollection({
|
||||
name: 'embedchain_store',
|
||||
embeddingFunction: embedder,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export { ChromaDB };
|
||||
@@ -1,3 +0,0 @@
|
||||
import { ChromaDB } from './ChromaDb';
|
||||
|
||||
export { ChromaDB };
|
||||
@@ -1,9 +0,0 @@
|
||||
const { EmbedChainApp } = require("./embedchain/embedchain");
|
||||
|
||||
async function App() {
|
||||
const app = new EmbedChainApp();
|
||||
await app.init_app;
|
||||
return app;
|
||||
}
|
||||
|
||||
module.exports = { App };
|
||||
@@ -1,5 +0,0 @@
|
||||
module.exports = {
|
||||
preset: 'ts-jest',
|
||||
testEnvironment: 'node',
|
||||
testPathIgnorePatterns: ['.d.ts'],
|
||||
};
|
||||
@@ -1,5 +0,0 @@
|
||||
module.exports = {
|
||||
'*.{js,ts}': ['eslint --fix', 'eslint'],
|
||||
'**/*.ts?(x)': () => 'npm run check-types',
|
||||
'*.json': ['prettier --write'],
|
||||
};
|
||||
Generated
-18457
File diff suppressed because it is too large
Load Diff
@@ -1,53 +0,0 @@
|
||||
{
|
||||
"name": "embedchain",
|
||||
"version": "0.0.8",
|
||||
"description": "embedchain is a framework to easily create LLM powered bots over any dataset",
|
||||
"main": "dist/index.js",
|
||||
"types": "types/index.d.ts",
|
||||
"files": [
|
||||
"dist",
|
||||
"types"
|
||||
],
|
||||
"scripts": {
|
||||
"build": "tsc -p tsconfig.build.json --listFiles",
|
||||
"prepare": "husky install",
|
||||
"test": "jest",
|
||||
"check-types": "tsc --noEmit --pretty"
|
||||
},
|
||||
"author": "Taranjeet Singh",
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"axios": "^1.4.0",
|
||||
"chromadb": "^1.5.6",
|
||||
"jsdom": "^22.1.0",
|
||||
"langchain": "^0.0.136",
|
||||
"openai": "^4.3.1",
|
||||
"pdfjs-dist": "^3.8.162",
|
||||
"uuid": "^9.0.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@commitlint/cli": "^17.1.2",
|
||||
"@commitlint/config-conventional": "^17.1.0",
|
||||
"@commitlint/cz-commitlint": "^17.1.2",
|
||||
"@types/jest": "^29.5.1",
|
||||
"@types/jsdom": "^21.1.1",
|
||||
"@typescript-eslint/eslint-plugin": "^5.41.0",
|
||||
"@typescript-eslint/parser": "^5.41.0",
|
||||
"eslint": "^8.34.0",
|
||||
"eslint-config-airbnb-base": "^15.0.0",
|
||||
"eslint-config-airbnb-typescript": "^17.0.0",
|
||||
"eslint-config-prettier": "^8.5.0",
|
||||
"eslint-plugin-import": "^2.27.5",
|
||||
"eslint-plugin-prettier": "^4.2.1",
|
||||
"eslint-plugin-simple-import-sort": "^8.0.0",
|
||||
"eslint-plugin-testing-library": "^5.9.1",
|
||||
"eslint-plugin-unused-imports": "^2.0.0",
|
||||
"husky": "^8.0.1",
|
||||
"jest": "^29.5.0",
|
||||
"lint-staged": "^13.0.3",
|
||||
"prettier": "^2.7.1",
|
||||
"ts-jest": "^29.1.0",
|
||||
"ts-loader": "^9.4.2",
|
||||
"typescript": "^5.2.2"
|
||||
}
|
||||
}
|
||||
@@ -1,4 +0,0 @@
|
||||
{
|
||||
"extends": "./tsconfig.json",
|
||||
"exclude": ["embedchain/__tests__"]
|
||||
}
|
||||
@@ -1,15 +0,0 @@
|
||||
{
|
||||
"compilerOptions": {
|
||||
"target": "es6",
|
||||
"module": "CommonJS",
|
||||
"strict": true,
|
||||
"outDir": "dist",
|
||||
"rootDir": "embedchain",
|
||||
"sourceMap": true,
|
||||
"declaration": true,
|
||||
"declarationDir": "types",
|
||||
"esModuleInterop": true
|
||||
},
|
||||
"include": ["embedchain/**/*.ts"],
|
||||
"exclude": ["node_modules", "dist"]
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
from typing import Optional
|
||||
|
||||
from langchain.text_splitter import RecursiveCharacterTextSplitter
|
||||
|
||||
from embedchain.chunkers.base_chunker import BaseChunker
|
||||
from embedchain.config.add_config import ChunkerConfig
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class AudioChunker(BaseChunker):
|
||||
"""Chunker for audio."""
|
||||
|
||||
def __init__(self, config: Optional[ChunkerConfig] = None):
|
||||
if config is None:
|
||||
config = ChunkerConfig(chunk_size=1000, chunk_overlap=0, length_function=len)
|
||||
text_splitter = RecursiveCharacterTextSplitter(
|
||||
chunk_size=config.chunk_size,
|
||||
chunk_overlap=config.chunk_overlap,
|
||||
length_function=config.length_function,
|
||||
)
|
||||
super().__init__(text_splitter)
|
||||
@@ -1,4 +1,4 @@
|
||||
from typing import Optional
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
|
||||
@@ -10,8 +10,10 @@ class BaseEmbedderConfig:
|
||||
model: Optional[str] = None,
|
||||
deployment_name: Optional[str] = None,
|
||||
vector_dimension: Optional[int] = None,
|
||||
endpoint: Optional[str] = None,
|
||||
api_key: Optional[str] = None,
|
||||
api_base: Optional[str] = None,
|
||||
model_kwargs: Optional[Dict[str, Any]] = None,
|
||||
):
|
||||
"""
|
||||
Initialize a new instance of an embedder config class.
|
||||
@@ -20,9 +22,21 @@ class BaseEmbedderConfig:
|
||||
:type model: Optional[str], optional
|
||||
:param deployment_name: deployment name for llm embedding model, defaults to None
|
||||
:type deployment_name: Optional[str], optional
|
||||
:param vector_dimension: vector dimension of the embedding model, defaults to None
|
||||
:type vector_dimension: Optional[int], optional
|
||||
:param endpoint: endpoint for the embedding model, defaults to None
|
||||
:type endpoint: Optional[str], optional
|
||||
:param api_key: hugginface api key, defaults to None
|
||||
:type api_key: Optional[str], optional
|
||||
:param api_base: huggingface api base, defaults to None
|
||||
:type api_base: Optional[str], optional
|
||||
:param model_kwargs: key-value arguments for the embedding model, defaults a dict inside init.
|
||||
:type model_kwargs: Optional[Dict[str, Any]], defaults a dict inside init.
|
||||
"""
|
||||
self.model = model
|
||||
self.deployment_name = deployment_name
|
||||
self.vector_dimension = vector_dimension
|
||||
self.endpoint = endpoint
|
||||
self.api_key = api_key
|
||||
self.api_base = api_base
|
||||
self.model_kwargs = model_kwargs or {}
|
||||
|
||||
@@ -10,9 +10,10 @@ class GoogleAIEmbedderConfig(BaseEmbedderConfig):
|
||||
self,
|
||||
model: Optional[str] = None,
|
||||
deployment_name: Optional[str] = None,
|
||||
vector_dimension: Optional[int] = None,
|
||||
task_type: Optional[str] = None,
|
||||
title: Optional[str] = None,
|
||||
):
|
||||
super().__init__(model, deployment_name)
|
||||
super().__init__(model, deployment_name, vector_dimension)
|
||||
self.task_type = task_type or "retrieval_document"
|
||||
self.title = title or "Embeddings for Embedchain"
|
||||
|
||||
@@ -10,6 +10,7 @@ class OllamaEmbedderConfig(BaseEmbedderConfig):
|
||||
self,
|
||||
model: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
vector_dimension: Optional[int] = None,
|
||||
):
|
||||
super().__init__(model)
|
||||
super().__init__(model=model, vector_dimension=vector_dimension)
|
||||
self.base_url = base_url or "http://localhost:11434"
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
import logging
|
||||
import re
|
||||
from string import Template
|
||||
from typing import Any, Mapping, Optional
|
||||
from typing import Any, Mapping, Optional, Dict, Union
|
||||
|
||||
import httpx
|
||||
|
||||
from embedchain.config.base_config import BaseConfig
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
@@ -99,10 +101,11 @@ class BaseLlmConfig(BaseConfig):
|
||||
base_url: Optional[str] = None,
|
||||
endpoint: Optional[str] = None,
|
||||
model_kwargs: Optional[dict[str, Any]] = None,
|
||||
http_client: Optional[Any] = None,
|
||||
http_async_client: Optional[Any] = None,
|
||||
http_client_proxies: Optional[Union[Dict, str]] = None,
|
||||
http_async_client_proxies: Optional[Union[Dict, str]] = None,
|
||||
local: Optional[bool] = False,
|
||||
default_headers: Optional[Mapping[str, str]] = None,
|
||||
api_version: Optional[str] = None,
|
||||
):
|
||||
"""
|
||||
Initializes a configuration class instance for the LLM.
|
||||
@@ -148,6 +151,11 @@ class BaseLlmConfig(BaseConfig):
|
||||
:type callbacks: Optional[list], optional
|
||||
:param query_type: The type of query to use, defaults to None
|
||||
:type query_type: Optional[str], optional
|
||||
:param http_client_proxies: The proxy server settings used to create self.http_client, defaults to None
|
||||
:type http_client_proxies: Optional[Dict | str], optional
|
||||
:param http_async_client_proxies: The proxy server settings for async calls used to create
|
||||
self.http_async_client, defaults to None
|
||||
:type http_async_client_proxies: Optional[Dict | str], optional
|
||||
:param local: If True, the model will be run locally, defaults to False (for huggingface provider)
|
||||
:type local: Optional[bool], optional
|
||||
:param default_headers: Set additional HTTP headers to be sent with requests to OpenAI
|
||||
@@ -180,11 +188,14 @@ class BaseLlmConfig(BaseConfig):
|
||||
self.base_url = base_url
|
||||
self.endpoint = endpoint
|
||||
self.model_kwargs = model_kwargs
|
||||
self.http_client = http_client
|
||||
self.http_async_client = http_async_client
|
||||
self.http_client = httpx.Client(proxies=http_client_proxies) if http_client_proxies else None
|
||||
self.http_async_client = (
|
||||
httpx.AsyncClient(proxies=http_async_client_proxies) if http_async_client_proxies else None
|
||||
)
|
||||
self.local = local
|
||||
self.default_headers = default_headers
|
||||
self.online = online
|
||||
self.api_version = api_version
|
||||
|
||||
if isinstance(prompt, str):
|
||||
prompt = Template(prompt)
|
||||
|
||||
@@ -12,6 +12,7 @@ class ChromaDbConfig(BaseVectorDbConfig):
|
||||
dir: Optional[str] = None,
|
||||
host: Optional[str] = None,
|
||||
port: Optional[str] = None,
|
||||
batch_size: Optional[int] = 100,
|
||||
allow_reset=False,
|
||||
chroma_settings: Optional[dict] = None,
|
||||
):
|
||||
@@ -26,6 +27,8 @@ class ChromaDbConfig(BaseVectorDbConfig):
|
||||
:type host: Optional[str], optional
|
||||
:param port: Database connection remote port. Use this if you run Embedchain as a client, defaults to None
|
||||
:type port: Optional[str], optional
|
||||
:param batch_size: Number of items to insert in one batch, defaults to 100
|
||||
:type batch_size: Optional[int], optional
|
||||
:param allow_reset: Resets the database. defaults to False
|
||||
:type allow_reset: bool
|
||||
:param chroma_settings: Chroma settings dict, defaults to None
|
||||
@@ -34,4 +37,5 @@ class ChromaDbConfig(BaseVectorDbConfig):
|
||||
|
||||
self.chroma_settings = chroma_settings
|
||||
self.allow_reset = allow_reset
|
||||
self.batch_size = batch_size
|
||||
super().__init__(collection_name=collection_name, dir=dir, host=host, port=port)
|
||||
|
||||
@@ -13,6 +13,7 @@ class ElasticsearchDBConfig(BaseVectorDbConfig):
|
||||
dir: Optional[str] = None,
|
||||
es_url: Union[str, list[str]] = None,
|
||||
cloud_id: Optional[str] = None,
|
||||
batch_size: Optional[int] = 100,
|
||||
**ES_EXTRA_PARAMS: dict[str, any],
|
||||
):
|
||||
"""
|
||||
@@ -24,6 +25,10 @@ class ElasticsearchDBConfig(BaseVectorDbConfig):
|
||||
:type dir: Optional[str], optional
|
||||
:param es_url: elasticsearch url or list of nodes url to be used for connection, defaults to None
|
||||
:type es_url: Union[str, list[str]], optional
|
||||
:param cloud_id: cloud id of the elasticsearch cluster, defaults to None
|
||||
:type cloud_id: Optional[str], optional
|
||||
:param batch_size: Number of items to insert in one batch, defaults to 100
|
||||
:type batch_size: Optional[int], optional
|
||||
:param ES_EXTRA_PARAMS: extra params dict that can be passed to elasticsearch.
|
||||
:type ES_EXTRA_PARAMS: dict[str, Any], optional
|
||||
"""
|
||||
@@ -46,4 +51,6 @@ class ElasticsearchDBConfig(BaseVectorDbConfig):
|
||||
and not self.ES_EXTRA_PARAMS.get("bearer_auth")
|
||||
):
|
||||
self.ES_EXTRA_PARAMS["api_key"] = os.environ.get("ELASTICSEARCH_API_KEY")
|
||||
|
||||
self.batch_size = batch_size
|
||||
super().__init__(collection_name=collection_name, dir=dir)
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
from typing import Optional
|
||||
|
||||
from embedchain.config.vectordb.base import BaseVectorDbConfig
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class LanceDBConfig(BaseVectorDbConfig):
|
||||
def __init__(
|
||||
self,
|
||||
collection_name: Optional[str] = None,
|
||||
dir: Optional[str] = None,
|
||||
host: Optional[str] = None,
|
||||
port: Optional[str] = None,
|
||||
allow_reset=True,
|
||||
):
|
||||
"""
|
||||
Initializes a configuration class instance for LanceDB.
|
||||
|
||||
:param collection_name: Default name for the collection, defaults to None
|
||||
:type collection_name: Optional[str], optional
|
||||
:param dir: Path to the database directory, where the database is stored, defaults to None
|
||||
:type dir: Optional[str], optional
|
||||
:param host: Database connection remote host. Use this if you run Embedchain as a client, defaults to None
|
||||
:type host: Optional[str], optional
|
||||
:param port: Database connection remote port. Use this if you run Embedchain as a client, defaults to None
|
||||
:type port: Optional[str], optional
|
||||
:param allow_reset: Resets the database. defaults to False
|
||||
:type allow_reset: bool
|
||||
"""
|
||||
|
||||
self.allow_reset = allow_reset
|
||||
super().__init__(collection_name=collection_name, dir=dir, host=host, port=port)
|
||||
@@ -13,6 +13,7 @@ class OpenSearchDBConfig(BaseVectorDbConfig):
|
||||
vector_dimension: int = 1536,
|
||||
collection_name: Optional[str] = None,
|
||||
dir: Optional[str] = None,
|
||||
batch_size: Optional[int] = 100,
|
||||
**extra_params: dict[str, any],
|
||||
):
|
||||
"""
|
||||
@@ -28,10 +29,13 @@ class OpenSearchDBConfig(BaseVectorDbConfig):
|
||||
:type vector_dimension: int, optional
|
||||
:param dir: Path to the database directory, where the database is stored, defaults to None
|
||||
:type dir: Optional[str], optional
|
||||
:param batch_size: Number of items to insert in one batch, defaults to 100
|
||||
:type batch_size: Optional[int], optional
|
||||
"""
|
||||
self.opensearch_url = opensearch_url
|
||||
self.http_auth = http_auth
|
||||
self.vector_dimension = vector_dimension
|
||||
self.extra_params = extra_params
|
||||
self.batch_size = batch_size
|
||||
|
||||
super().__init__(collection_name=collection_name, dir=dir)
|
||||
|
||||
@@ -17,6 +17,7 @@ class PineconeDBConfig(BaseVectorDbConfig):
|
||||
serverless_config: Optional[dict[str, any]] = None,
|
||||
hybrid_search: bool = False,
|
||||
bm25_encoder: any = None,
|
||||
batch_size: Optional[int] = 100,
|
||||
**extra_params: dict[str, any],
|
||||
):
|
||||
self.metric = metric
|
||||
@@ -26,6 +27,7 @@ class PineconeDBConfig(BaseVectorDbConfig):
|
||||
self.extra_params = extra_params
|
||||
self.hybrid_search = hybrid_search
|
||||
self.bm25_encoder = bm25_encoder
|
||||
self.batch_size = batch_size
|
||||
if pod_config is None and serverless_config is None:
|
||||
# If no config is provided, use the default pod spec config
|
||||
pod_environment = os.environ.get("PINECONE_ENV", "gcp-starter")
|
||||
|
||||
@@ -18,6 +18,7 @@ class QdrantDBConfig(BaseVectorDbConfig):
|
||||
hnsw_config: Optional[dict[str, any]] = None,
|
||||
quantization_config: Optional[dict[str, any]] = None,
|
||||
on_disk: Optional[bool] = None,
|
||||
batch_size: Optional[int] = 10,
|
||||
**extra_params: dict[str, any],
|
||||
):
|
||||
"""
|
||||
@@ -36,9 +37,12 @@ class QdrantDBConfig(BaseVectorDbConfig):
|
||||
This setting saves RAM by (slightly) increasing the response time.
|
||||
Note: those payload values that are involved in filtering and are indexed - remain in RAM.
|
||||
:type on_disk: bool, optional, defaults to None
|
||||
:param batch_size: Number of items to insert in one batch, defaults to 10
|
||||
:type batch_size: Optional[int], optional
|
||||
"""
|
||||
self.hnsw_config = hnsw_config
|
||||
self.quantization_config = quantization_config
|
||||
self.on_disk = on_disk
|
||||
self.batch_size = batch_size
|
||||
self.extra_params = extra_params
|
||||
super().__init__(collection_name=collection_name, dir=dir)
|
||||
|
||||
@@ -10,7 +10,9 @@ class WeaviateDBConfig(BaseVectorDbConfig):
|
||||
self,
|
||||
collection_name: Optional[str] = None,
|
||||
dir: Optional[str] = None,
|
||||
batch_size: Optional[int] = 100,
|
||||
**extra_params: dict[str, any],
|
||||
):
|
||||
self.batch_size = batch_size
|
||||
self.extra_params = extra_params
|
||||
super().__init__(collection_name=collection_name, dir=dir)
|
||||
|
||||
@@ -81,6 +81,7 @@ class DataFormatter(JSONSerializable):
|
||||
DataType.DROPBOX: "embedchain.loaders.dropbox.DropboxLoader",
|
||||
DataType.TEXT_FILE: "embedchain.loaders.text_file.TextFileLoader",
|
||||
DataType.EXCEL_FILE: "embedchain.loaders.excel_file.ExcelFileLoader",
|
||||
DataType.AUDIO: "embedchain.loaders.audio.AudioLoader",
|
||||
}
|
||||
|
||||
if data_type == DataType.CUSTOM or loader is not None:
|
||||
@@ -129,6 +130,7 @@ class DataFormatter(JSONSerializable):
|
||||
DataType.DROPBOX: "embedchain.chunkers.common_chunker.CommonChunker",
|
||||
DataType.TEXT_FILE: "embedchain.chunkers.common_chunker.CommonChunker",
|
||||
DataType.EXCEL_FILE: "embedchain.chunkers.excel_file.ExcelFileChunker",
|
||||
DataType.AUDIO: "embedchain.chunkers.audio.AudioChunker",
|
||||
}
|
||||
|
||||
if chunker is not None:
|
||||
|
||||
@@ -6,7 +6,9 @@ from typing import Any, Optional, Union
|
||||
from dotenv import load_dotenv
|
||||
from langchain.docstore.document import Document
|
||||
|
||||
from embedchain.cache import adapt, get_gptcache_session, gptcache_data_convert, gptcache_update_cache_callback
|
||||
from embedchain.cache import (adapt, get_gptcache_session,
|
||||
gptcache_data_convert,
|
||||
gptcache_update_cache_callback)
|
||||
from embedchain.chunkers.base_chunker import BaseChunker
|
||||
from embedchain.config import AddConfig, BaseLlmConfig, ChunkerConfig
|
||||
from embedchain.config.base_app_config import BaseAppConfig
|
||||
@@ -16,7 +18,8 @@ from embedchain.embedder.base import BaseEmbedder
|
||||
from embedchain.helpers.json_serializable import JSONSerializable
|
||||
from embedchain.llm.base import BaseLlm
|
||||
from embedchain.loaders.base_loader import BaseLoader
|
||||
from embedchain.models.data_type import DataType, DirectDataType, IndirectDataType, SpecialDataType
|
||||
from embedchain.models.data_type import (DataType, DirectDataType,
|
||||
IndirectDataType, SpecialDataType)
|
||||
from embedchain.utils.misc import detect_datatype, is_valid_json_string
|
||||
from embedchain.vectordb.base import BaseVectorDB
|
||||
|
||||
|
||||
Binary file not shown.
@@ -0,0 +1,22 @@
|
||||
from typing import Optional
|
||||
|
||||
from langchain_community.embeddings import AzureOpenAIEmbeddings
|
||||
|
||||
from embedchain.config import BaseEmbedderConfig
|
||||
from embedchain.embedder.base import BaseEmbedder
|
||||
from embedchain.models import VectorDimensions
|
||||
|
||||
|
||||
class AzureOpenAIEmbedder(BaseEmbedder):
|
||||
def __init__(self, config: Optional[BaseEmbedderConfig] = None):
|
||||
super().__init__(config=config)
|
||||
|
||||
if self.config.model is None:
|
||||
self.config.model = "text-embedding-ada-002"
|
||||
|
||||
embeddings = AzureOpenAIEmbeddings(deployment=self.config.deployment_name)
|
||||
embedding_fn = BaseEmbedder._langchain_default_concept(embeddings)
|
||||
|
||||
self.set_embedding_fn(embedding_fn=embedding_fn)
|
||||
vector_dimension = self.config.vector_dimension or VectorDimensions.OPENAI.value
|
||||
self.set_vector_dimension(vector_dimension=vector_dimension)
|
||||
@@ -0,0 +1,52 @@
|
||||
import os
|
||||
from typing import Optional, Union
|
||||
|
||||
from embedchain.config import BaseEmbedderConfig
|
||||
from embedchain.embedder.base import BaseEmbedder
|
||||
|
||||
from chromadb import EmbeddingFunction, Embeddings
|
||||
|
||||
|
||||
class ClarifaiEmbeddingFunction(EmbeddingFunction):
|
||||
def __init__(self, config: BaseEmbedderConfig) -> None:
|
||||
super().__init__()
|
||||
try:
|
||||
from clarifai.client.model import Model
|
||||
from clarifai.client.input import Inputs
|
||||
except ModuleNotFoundError:
|
||||
raise ModuleNotFoundError(
|
||||
"The required dependencies for ClarifaiEmbeddingFunction are not installed."
|
||||
'Please install with `pip install --upgrade "embedchain[clarifai]"`'
|
||||
) from None
|
||||
self.config = config
|
||||
self.api_key = config.api_key or os.getenv("CLARIFAI_PAT")
|
||||
self.model = config.model
|
||||
self.model_obj = Model(url=self.model, pat=self.api_key)
|
||||
self.input_obj = Inputs(pat=self.api_key)
|
||||
|
||||
def __call__(self, input: Union[str, list[str]]) -> Embeddings:
|
||||
if isinstance(input, str):
|
||||
input = [input]
|
||||
|
||||
batch_size = 32
|
||||
embeddings = []
|
||||
try:
|
||||
for i in range(0, len(input), batch_size):
|
||||
batch = input[i : i + batch_size]
|
||||
input_batch = [
|
||||
self.input_obj.get_text_input(input_id=str(id), raw_text=inp) for id, inp in enumerate(batch)
|
||||
]
|
||||
response = self.model_obj.predict(input_batch)
|
||||
embeddings.extend([list(output.data.embeddings[0].vector) for output in response.outputs])
|
||||
except Exception as e:
|
||||
print(f"Predict failed, exception: {e}")
|
||||
|
||||
return embeddings
|
||||
|
||||
|
||||
class ClarifaiEmbedder(BaseEmbedder):
|
||||
def __init__(self, config: Optional[BaseEmbedderConfig] = None):
|
||||
super().__init__(config)
|
||||
|
||||
embedding_func = ClarifaiEmbeddingFunction(config=self.config)
|
||||
self.set_embedding_fn(embedding_fn=embedding_func)
|
||||
@@ -1,7 +1,16 @@
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
from langchain_community.embeddings import HuggingFaceEmbeddings
|
||||
|
||||
try:
|
||||
from langchain_huggingface import HuggingFaceEndpointEmbeddings
|
||||
except ModuleNotFoundError:
|
||||
raise ModuleNotFoundError(
|
||||
"The required dependencies for HuggingFaceHub are not installed."
|
||||
"Please install with `pip install langchain_huggingface`"
|
||||
) from None
|
||||
|
||||
from embedchain.config import BaseEmbedderConfig
|
||||
from embedchain.embedder.base import BaseEmbedder
|
||||
from embedchain.models import VectorDimensions
|
||||
@@ -11,7 +20,19 @@ class HuggingFaceEmbedder(BaseEmbedder):
|
||||
def __init__(self, config: Optional[BaseEmbedderConfig] = None):
|
||||
super().__init__(config=config)
|
||||
|
||||
embeddings = HuggingFaceEmbeddings(model_name=self.config.model)
|
||||
if self.config.endpoint:
|
||||
if not self.config.api_key and "HUGGINGFACE_ACCESS_TOKEN" not in os.environ:
|
||||
raise ValueError(
|
||||
"Please set the HUGGINGFACE_ACCESS_TOKEN environment variable or pass API Key in the config."
|
||||
)
|
||||
|
||||
embeddings = HuggingFaceEndpointEmbeddings(
|
||||
model=self.config.endpoint,
|
||||
huggingfacehub_api_token=self.config.api_key or os.getenv("HUGGINGFACE_ACCESS_TOKEN"),
|
||||
)
|
||||
else:
|
||||
embeddings = HuggingFaceEmbeddings(model_name=self.config.model, model_kwargs=self.config.model_kwargs)
|
||||
|
||||
embedding_fn = BaseEmbedder._langchain_default_concept(embeddings)
|
||||
self.set_embedding_fn(embedding_fn=embedding_fn)
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@ import os
|
||||
from typing import Optional
|
||||
|
||||
from chromadb.utils.embedding_functions import OpenAIEmbeddingFunction
|
||||
from langchain_community.embeddings import AzureOpenAIEmbeddings
|
||||
|
||||
|
||||
from embedchain.config import BaseEmbedderConfig
|
||||
from embedchain.embedder.base import BaseEmbedder
|
||||
@@ -19,20 +19,14 @@ class OpenAIEmbedder(BaseEmbedder):
|
||||
api_key = self.config.api_key or os.environ["OPENAI_API_KEY"]
|
||||
api_base = self.config.api_base or os.environ.get("OPENAI_API_BASE")
|
||||
|
||||
if self.config.deployment_name:
|
||||
embeddings = AzureOpenAIEmbeddings(deployment=self.config.deployment_name)
|
||||
embedding_fn = BaseEmbedder._langchain_default_concept(embeddings)
|
||||
else:
|
||||
if api_key is None and os.getenv("OPENAI_ORGANIZATION") is None:
|
||||
raise ValueError(
|
||||
"OPENAI_API_KEY or OPENAI_ORGANIZATION environment variables not provided"
|
||||
) # noqa:E501
|
||||
embedding_fn = OpenAIEmbeddingFunction(
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
organization_id=os.getenv("OPENAI_ORGANIZATION"),
|
||||
model_name=self.config.model,
|
||||
)
|
||||
if api_key is None and os.getenv("OPENAI_ORGANIZATION") is None:
|
||||
raise ValueError("OPENAI_API_KEY or OPENAI_ORGANIZATION environment variables not provided") # noqa:E501
|
||||
embedding_fn = OpenAIEmbeddingFunction(
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
organization_id=os.getenv("OPENAI_ORGANIZATION"),
|
||||
model_name=self.config.model,
|
||||
)
|
||||
self.set_embedding_fn(embedding_fn=embedding_fn)
|
||||
vector_dimension = self.config.vector_dimension or VectorDimensions.OPENAI.value
|
||||
self.set_vector_dimension(vector_dimension=vector_dimension)
|
||||
|
||||
@@ -23,6 +23,7 @@ class LlmFactory:
|
||||
"google": "embedchain.llm.google.GoogleLlm",
|
||||
"aws_bedrock": "embedchain.llm.aws_bedrock.AWSBedrockLlm",
|
||||
"mistralai": "embedchain.llm.mistralai.MistralAILlm",
|
||||
"clarifai": "embedchain.llm.clarifai.ClarifaiLlm",
|
||||
"groq": "embedchain.llm.groq.GroqLlm",
|
||||
"nvidia": "embedchain.llm.nvidia.NvidiaLlm",
|
||||
"vllm": "embedchain.llm.vllm.VLLM",
|
||||
@@ -49,13 +50,14 @@ class LlmFactory:
|
||||
|
||||
class EmbedderFactory:
|
||||
provider_to_class = {
|
||||
"azure_openai": "embedchain.embedder.openai.OpenAIEmbedder",
|
||||
"azure_openai": "embedchain.embedder.azure_openai.AzureOpenAIEmbedder",
|
||||
"gpt4all": "embedchain.embedder.gpt4all.GPT4AllEmbedder",
|
||||
"huggingface": "embedchain.embedder.huggingface.HuggingFaceEmbedder",
|
||||
"openai": "embedchain.embedder.openai.OpenAIEmbedder",
|
||||
"vertexai": "embedchain.embedder.vertexai.VertexAIEmbedder",
|
||||
"google": "embedchain.embedder.google.GoogleAIEmbedder",
|
||||
"mistralai": "embedchain.embedder.mistralai.MistralAIEmbedder",
|
||||
"clarifai": "embedchain.embedder.clarifai.ClarifaiEmbedder",
|
||||
"nvidia": "embedchain.embedder.nvidia.NvidiaEmbedder",
|
||||
"cohere": "embedchain.embedder.cohere.CohereEmbedder",
|
||||
"ollama": "embedchain.embedder.ollama.OllamaEmbedder",
|
||||
@@ -65,6 +67,7 @@ class EmbedderFactory:
|
||||
"google": "embedchain.config.embedder.google.GoogleAIEmbedderConfig",
|
||||
"gpt4all": "embedchain.config.embedder.base.BaseEmbedderConfig",
|
||||
"huggingface": "embedchain.config.embedder.base.BaseEmbedderConfig",
|
||||
"clarifai": "embedchain.config.embedder.base.BaseEmbedderConfig",
|
||||
"openai": "embedchain.config.embedder.base.BaseEmbedderConfig",
|
||||
"ollama": "embedchain.config.embedder.ollama.OllamaEmbedderConfig",
|
||||
}
|
||||
@@ -88,6 +91,7 @@ class VectorDBFactory:
|
||||
"chroma": "embedchain.vectordb.chroma.ChromaDB",
|
||||
"elasticsearch": "embedchain.vectordb.elasticsearch.ElasticsearchDB",
|
||||
"opensearch": "embedchain.vectordb.opensearch.OpenSearchDB",
|
||||
"lancedb": "embedchain.vectordb.lancedb.LanceDB",
|
||||
"pinecone": "embedchain.vectordb.pinecone.PineconeDB",
|
||||
"qdrant": "embedchain.vectordb.qdrant.QdrantDB",
|
||||
"weaviate": "embedchain.vectordb.weaviate.WeaviateDB",
|
||||
@@ -97,6 +101,7 @@ class VectorDBFactory:
|
||||
"chroma": "embedchain.config.vectordb.chroma.ChromaDbConfig",
|
||||
"elasticsearch": "embedchain.config.vectordb.elasticsearch.ElasticsearchDBConfig",
|
||||
"opensearch": "embedchain.config.vectordb.opensearch.OpenSearchDBConfig",
|
||||
"lancedb": "embedchain.config.vectordb.lancedb.LanceDBConfig",
|
||||
"pinecone": "embedchain.config.vectordb.pinecone.PineconeDBConfig",
|
||||
"qdrant": "embedchain.config.vectordb.qdrant.QdrantDBConfig",
|
||||
"weaviate": "embedchain.config.vectordb.weaviate.WeaviateDBConfig",
|
||||
|
||||
@@ -17,18 +17,17 @@ logger = logging.getLogger(__name__)
|
||||
@register_deserializable
|
||||
class AnthropicLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
if "ANTHROPIC_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the ANTHROPIC_API_KEY environment variable.")
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "ANTHROPIC_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the ANTHROPIC_API_KEY environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
return AnthropicLlm._get_answer(prompt=prompt, config=self.config)
|
||||
|
||||
@staticmethod
|
||||
def _get_answer(prompt: str, config: BaseLlmConfig) -> str:
|
||||
chat = ChatAnthropic(
|
||||
anthropic_api_key=os.environ["ANTHROPIC_API_KEY"], temperature=config.temperature, model_name=config.model
|
||||
)
|
||||
api_key = config.api_key or os.getenv("ANTHROPIC_API_KEY")
|
||||
chat = ChatAnthropic(anthropic_api_key=api_key, temperature=config.temperature, model_name=config.model)
|
||||
|
||||
if config.max_tokens and config.max_tokens != 1000:
|
||||
logger.warning("Config option `max_tokens` is not supported by this model.")
|
||||
|
||||
@@ -14,18 +14,18 @@ class AzureOpenAILlm(BaseLlm):
|
||||
super().__init__(config=config)
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
return AzureOpenAILlm._get_answer(prompt=prompt, config=self.config)
|
||||
return self._get_answer(prompt=prompt, config=self.config)
|
||||
|
||||
@staticmethod
|
||||
def _get_answer(prompt: str, config: BaseLlmConfig) -> str:
|
||||
from langchain_community.chat_models import AzureChatOpenAI
|
||||
from langchain_openai import AzureChatOpenAI
|
||||
|
||||
if not config.deployment_name:
|
||||
raise ValueError("Deployment name must be provided for Azure OpenAI")
|
||||
|
||||
chat = AzureChatOpenAI(
|
||||
deployment_name=config.deployment_name,
|
||||
openai_api_version="2023-05-15",
|
||||
openai_api_version=str(config.api_version) if config.api_version else "2024-02-01",
|
||||
model_name=config.model or "gpt-3.5-turbo",
|
||||
temperature=config.temperature,
|
||||
max_tokens=config.max_tokens,
|
||||
@@ -37,4 +37,4 @@ class AzureOpenAILlm(BaseLlm):
|
||||
|
||||
messages = BaseLlm._get_messages(prompt, system_prompt=config.system_prompt)
|
||||
|
||||
return chat(messages).content
|
||||
return chat.invoke(messages).content
|
||||
|
||||
@@ -5,7 +5,9 @@ from typing import Any, Optional
|
||||
from langchain.schema import BaseMessage as LCBaseMessage
|
||||
|
||||
from embedchain.config import BaseLlmConfig
|
||||
from embedchain.config.llm.base import DEFAULT_PROMPT, DEFAULT_PROMPT_WITH_HISTORY_TEMPLATE, DOCS_SITE_PROMPT_TEMPLATE
|
||||
from embedchain.config.llm.base import (DEFAULT_PROMPT,
|
||||
DEFAULT_PROMPT_WITH_HISTORY_TEMPLATE,
|
||||
DOCS_SITE_PROMPT_TEMPLATE)
|
||||
from embedchain.helpers.json_serializable import JSONSerializable
|
||||
from embedchain.memory.base import ChatHistory
|
||||
from embedchain.memory.message import ChatMessage
|
||||
|
||||
@@ -0,0 +1,47 @@
|
||||
import logging
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
from embedchain.config import BaseLlmConfig
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
from embedchain.llm.base import BaseLlm
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class ClarifaiLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "CLARIFAI_PAT" not in os.environ:
|
||||
raise ValueError("Please set the CLARIFAI_PAT environment variable.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
return self._get_answer(prompt=prompt, config=self.config)
|
||||
|
||||
@staticmethod
|
||||
def _get_answer(prompt: str, config: BaseLlmConfig) -> str:
|
||||
try:
|
||||
from clarifai.client.model import Model
|
||||
except ModuleNotFoundError:
|
||||
raise ModuleNotFoundError(
|
||||
"The required dependencies for Clarifai are not installed."
|
||||
'Please install with `pip install --upgrade "embedchain[clarifai]"`'
|
||||
) from None
|
||||
|
||||
model_name = config.model
|
||||
logging.info(f"Using clarifai LLM model: {model_name}")
|
||||
api_key = config.api_key or os.getenv("CLARIFAI_PAT")
|
||||
model = Model(url=model_name, pat=api_key)
|
||||
params = config.model_kwargs
|
||||
|
||||
try:
|
||||
(params := {}) if config.model_kwargs is None else config.model_kwargs
|
||||
predict_response = model.predict_by_bytes(
|
||||
bytes(prompt, "utf-8"),
|
||||
input_type="text",
|
||||
inference_params=params,
|
||||
)
|
||||
text = predict_response.outputs[0].data.text.raw
|
||||
return text
|
||||
|
||||
except Exception as e:
|
||||
logging.error(f"Predict failed, exception: {e}")
|
||||
@@ -12,9 +12,6 @@ from embedchain.llm.base import BaseLlm
|
||||
@register_deserializable
|
||||
class CohereLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
if "COHERE_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the COHERE_API_KEY environment variable.")
|
||||
|
||||
try:
|
||||
importlib.import_module("cohere")
|
||||
except ModuleNotFoundError:
|
||||
@@ -24,6 +21,8 @@ class CohereLlm(BaseLlm):
|
||||
) from None
|
||||
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "COHERE_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the COHERE_API_KEY environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
if self.config.system_prompt:
|
||||
@@ -32,8 +31,9 @@ class CohereLlm(BaseLlm):
|
||||
|
||||
@staticmethod
|
||||
def _get_answer(prompt: str, config: BaseLlmConfig) -> str:
|
||||
api_key = config.api_key or os.getenv("COHERE_API_KEY")
|
||||
llm = Cohere(
|
||||
cohere_api_key=os.environ["COHERE_API_KEY"],
|
||||
cohere_api_key=api_key,
|
||||
model=config.model,
|
||||
max_tokens=config.max_tokens,
|
||||
temperature=config.temperature,
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
import importlib
|
||||
import logging
|
||||
import os
|
||||
from collections.abc import Generator
|
||||
from typing import Any, Optional, Union
|
||||
|
||||
import google.generativeai as genai
|
||||
try:
|
||||
import google.generativeai as genai
|
||||
except ImportError:
|
||||
raise ImportError("GoogleLlm requires extra dependencies. Install with `pip install google-generativeai`") from None
|
||||
|
||||
from embedchain.config import BaseLlmConfig
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
@@ -16,19 +18,12 @@ logger = logging.getLogger(__name__)
|
||||
@register_deserializable
|
||||
class GoogleLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
if "GOOGLE_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the GOOGLE_API_KEY environment variable.")
|
||||
|
||||
try:
|
||||
importlib.import_module("google.generativeai")
|
||||
except ModuleNotFoundError:
|
||||
raise ModuleNotFoundError(
|
||||
"The required dependencies for GoogleLlm are not installed."
|
||||
'Please install with `pip install --upgrade "embedchain[google]"`'
|
||||
) from None
|
||||
|
||||
super().__init__(config)
|
||||
genai.configure(api_key=os.environ["GOOGLE_API_KEY"])
|
||||
if not self.config.api_key and "GOOGLE_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the GOOGLE_API_KEY environment variable or pass it in the config.")
|
||||
|
||||
api_key = self.config.api_key or os.getenv("GOOGLE_API_KEY")
|
||||
genai.configure(api_key=api_key)
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
if self.config.system_prompt:
|
||||
|
||||
@@ -19,6 +19,8 @@ from embedchain.llm.base import BaseLlm
|
||||
class GroqLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "GROQ_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the GROQ_API_KEY environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt) -> str:
|
||||
response = self._get_answer(prompt, self.config)
|
||||
|
||||
@@ -17,9 +17,6 @@ logger = logging.getLogger(__name__)
|
||||
@register_deserializable
|
||||
class HuggingFaceLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
if "HUGGINGFACE_ACCESS_TOKEN" not in os.environ:
|
||||
raise ValueError("Please set the HUGGINGFACE_ACCESS_TOKEN environment variable.")
|
||||
|
||||
try:
|
||||
importlib.import_module("huggingface_hub")
|
||||
except ModuleNotFoundError:
|
||||
@@ -29,6 +26,8 @@ class HuggingFaceLlm(BaseLlm):
|
||||
) from None
|
||||
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "HUGGINGFACE_ACCESS_TOKEN" not in os.environ:
|
||||
raise ValueError("Please set the HUGGINGFACE_ACCESS_TOKEN environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
if self.config.system_prompt:
|
||||
@@ -60,9 +59,10 @@ class HuggingFaceLlm(BaseLlm):
|
||||
raise ValueError("`top_p` must be > 0.0 and < 1.0")
|
||||
|
||||
model = config.model
|
||||
api_key = config.api_key or os.getenv("HUGGINGFACE_ACCESS_TOKEN")
|
||||
logger.info(f"Using HuggingFaceHub with model {model}")
|
||||
llm = HuggingFaceHub(
|
||||
huggingfacehub_api_token=os.environ["HUGGINGFACE_ACCESS_TOKEN"],
|
||||
huggingfacehub_api_token=api_key,
|
||||
repo_id=model,
|
||||
model_kwargs=model_kwargs,
|
||||
)
|
||||
@@ -70,8 +70,9 @@ class HuggingFaceLlm(BaseLlm):
|
||||
|
||||
@staticmethod
|
||||
def _from_endpoint(prompt: str, config: BaseLlmConfig) -> str:
|
||||
api_key = config.api_key or os.getenv("HUGGINGFACE_ACCESS_TOKEN")
|
||||
llm = HuggingFaceEndpoint(
|
||||
huggingfacehub_api_token=os.environ["HUGGINGFACE_ACCESS_TOKEN"],
|
||||
huggingfacehub_api_token=api_key,
|
||||
endpoint_url=config.endpoint,
|
||||
task="text-generation",
|
||||
model_kwargs=config.model_kwargs,
|
||||
|
||||
@@ -12,9 +12,9 @@ from embedchain.llm.base import BaseLlm
|
||||
@register_deserializable
|
||||
class JinaLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
if "JINACHAT_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the JINACHAT_API_KEY environment variable.")
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "JINACHAT_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the JINACHAT_API_KEY environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
response = JinaLlm._get_answer(prompt, self.config)
|
||||
@@ -29,6 +29,7 @@ class JinaLlm(BaseLlm):
|
||||
kwargs = {
|
||||
"temperature": config.temperature,
|
||||
"max_tokens": config.max_tokens,
|
||||
"jinachat_api_key": config.api_key or os.environ["JINACHAT_API_KEY"],
|
||||
"model_kwargs": {},
|
||||
}
|
||||
if config.top_p:
|
||||
|
||||
@@ -19,8 +19,6 @@ class Llama2Llm(BaseLlm):
|
||||
"The required dependencies for Llama2 are not installed."
|
||||
'Please install with `pip install --upgrade "embedchain[llama2]"`'
|
||||
) from None
|
||||
if "REPLICATE_API_TOKEN" not in os.environ:
|
||||
raise ValueError("Please set the REPLICATE_API_TOKEN environment variable.")
|
||||
|
||||
# Set default config values specific to this llm
|
||||
if not config:
|
||||
@@ -35,13 +33,17 @@ class Llama2Llm(BaseLlm):
|
||||
)
|
||||
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "REPLICATE_API_TOKEN" not in os.environ:
|
||||
raise ValueError("Please set the REPLICATE_API_TOKEN environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
# TODO: Move the model and other inputs into config
|
||||
if self.config.system_prompt:
|
||||
raise ValueError("Llama2 does not support `system_prompt`")
|
||||
api_key = self.config.api_key or os.getenv("REPLICATE_API_TOKEN")
|
||||
llm = Replicate(
|
||||
model=self.config.model,
|
||||
replicate_api_token=api_key,
|
||||
input={
|
||||
"temperature": self.config.temperature,
|
||||
"max_length": self.config.max_tokens,
|
||||
|
||||
@@ -21,10 +21,9 @@ from embedchain.llm.base import BaseLlm
|
||||
@register_deserializable
|
||||
class NvidiaLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
if "NVIDIA_API_KEY" not in os.environ:
|
||||
raise ValueError("NVIDIA_API_KEY environment variable must be set")
|
||||
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "NVIDIA_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the NVIDIA_API_KEY environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
return self._get_answer(prompt=prompt, config=self.config)
|
||||
@@ -34,7 +33,7 @@ class NvidiaLlm(BaseLlm):
|
||||
callback_manager = [StreamingStdOutCallbackHandler()] if config.stream else [StdOutCallbackHandler()]
|
||||
model_kwargs = config.model_kwargs or {}
|
||||
labels = model_kwargs.get("labels", None)
|
||||
params = {"model": config.model}
|
||||
params = {"model": config.model, "nvidia_api_key": config.api_key or os.getenv("NVIDIA_API_KEY")}
|
||||
if config.system_prompt:
|
||||
params["system_prompt"] = config.system_prompt
|
||||
if config.temperature:
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import logging
|
||||
from collections.abc import Iterable
|
||||
from typing import Optional, Union
|
||||
|
||||
@@ -6,10 +7,17 @@ from langchain.callbacks.stdout import StdOutCallbackHandler
|
||||
from langchain.callbacks.streaming_stdout import StreamingStdOutCallbackHandler
|
||||
from langchain_community.llms.ollama import Ollama
|
||||
|
||||
try:
|
||||
from ollama import Client
|
||||
except ImportError:
|
||||
raise ImportError("Ollama requires extra dependencies. Install with `pip install ollama`") from None
|
||||
|
||||
from embedchain.config import BaseLlmConfig
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
from embedchain.llm.base import BaseLlm
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class OllamaLlm(BaseLlm):
|
||||
@@ -18,19 +26,28 @@ class OllamaLlm(BaseLlm):
|
||||
if self.config.model is None:
|
||||
self.config.model = "llama2"
|
||||
|
||||
client = Client(host=config.base_url)
|
||||
local_models = client.list()["models"]
|
||||
if not any(model.get("name") == self.config.model for model in local_models):
|
||||
logger.info(f"Pulling {self.config.model} from Ollama!")
|
||||
client.pull(self.config.model)
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
return self._get_answer(prompt=prompt, config=self.config)
|
||||
|
||||
@staticmethod
|
||||
def _get_answer(prompt: str, config: BaseLlmConfig) -> Union[str, Iterable]:
|
||||
callback_manager = [StreamingStdOutCallbackHandler()] if config.stream else [StdOutCallbackHandler()]
|
||||
if config.stream:
|
||||
callbacks = config.callbacks if config.callbacks else [StreamingStdOutCallbackHandler()]
|
||||
else:
|
||||
callbacks = [StdOutCallbackHandler()]
|
||||
|
||||
llm = Ollama(
|
||||
model=config.model,
|
||||
system=config.system_prompt,
|
||||
temperature=config.temperature,
|
||||
top_p=config.top_p,
|
||||
callback_manager=CallbackManager(callback_manager),
|
||||
callback_manager=CallbackManager(callbacks),
|
||||
base_url=config.base_url,
|
||||
)
|
||||
|
||||
|
||||
@@ -36,7 +36,7 @@ class OpenAILlm(BaseLlm):
|
||||
"model": config.model or "gpt-3.5-turbo",
|
||||
"temperature": config.temperature,
|
||||
"max_tokens": config.max_tokens,
|
||||
"model_kwargs": {},
|
||||
"model_kwargs": config.model_kwargs or {},
|
||||
}
|
||||
api_key = config.api_key or os.environ["OPENAI_API_KEY"]
|
||||
base_url = config.base_url or os.environ.get("OPENAI_API_BASE", None)
|
||||
@@ -56,7 +56,13 @@ class OpenAILlm(BaseLlm):
|
||||
http_async_client=config.http_async_client,
|
||||
)
|
||||
else:
|
||||
chat = ChatOpenAI(**kwargs, api_key=api_key, base_url=base_url)
|
||||
chat = ChatOpenAI(
|
||||
**kwargs,
|
||||
api_key=api_key,
|
||||
base_url=base_url,
|
||||
http_client=config.http_client,
|
||||
http_async_client=config.http_async_client,
|
||||
)
|
||||
if self.tools:
|
||||
return self._query_function_call(chat, self.tools, messages)
|
||||
|
||||
|
||||
@@ -12,9 +12,6 @@ from embedchain.llm.base import BaseLlm
|
||||
@register_deserializable
|
||||
class TogetherLlm(BaseLlm):
|
||||
def __init__(self, config: Optional[BaseLlmConfig] = None):
|
||||
if "TOGETHER_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the TOGETHER_API_KEY environment variable.")
|
||||
|
||||
try:
|
||||
importlib.import_module("together")
|
||||
except ModuleNotFoundError:
|
||||
@@ -24,6 +21,8 @@ class TogetherLlm(BaseLlm):
|
||||
) from None
|
||||
|
||||
super().__init__(config=config)
|
||||
if not self.config.api_key and "TOGETHER_API_KEY" not in os.environ:
|
||||
raise ValueError("Please set the TOGETHER_API_KEY environment variable or pass it in the config.")
|
||||
|
||||
def get_llm_model_answer(self, prompt):
|
||||
if self.config.system_prompt:
|
||||
@@ -32,8 +31,9 @@ class TogetherLlm(BaseLlm):
|
||||
|
||||
@staticmethod
|
||||
def _get_answer(prompt: str, config: BaseLlmConfig) -> str:
|
||||
api_key = config.api_key or os.getenv("TOGETHER_API_KEY")
|
||||
llm = Together(
|
||||
together_api_key=os.environ["TOGETHER_API_KEY"],
|
||||
together_api_key=api_key,
|
||||
model=config.model,
|
||||
max_tokens=config.max_tokens,
|
||||
temperature=config.temperature,
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
import hashlib
|
||||
import os
|
||||
|
||||
import validators
|
||||
|
||||
from embedchain.helpers.json_serializable import register_deserializable
|
||||
from embedchain.loaders.base_loader import BaseLoader
|
||||
|
||||
try:
|
||||
from deepgram import DeepgramClient, PrerecordedOptions
|
||||
except ImportError:
|
||||
raise ImportError(
|
||||
"Audio file requires extra dependencies. Install with `pip install deepgram-sdk==3.2.7`"
|
||||
) from None
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class AudioLoader(BaseLoader):
|
||||
def __init__(self):
|
||||
if not os.environ.get("DEEPGRAM_API_KEY"):
|
||||
raise ValueError("DEEPGRAM_API_KEY is not set")
|
||||
|
||||
DG_KEY = os.environ.get("DEEPGRAM_API_KEY")
|
||||
self.client = DeepgramClient(DG_KEY)
|
||||
|
||||
def load_data(self, url: str):
|
||||
"""Load data from a audio file or URL."""
|
||||
|
||||
options = PrerecordedOptions(
|
||||
model="nova-2",
|
||||
smart_format=True,
|
||||
)
|
||||
if validators.url(url):
|
||||
source = {"url": url}
|
||||
response = self.client.listen.prerecorded.v("1").transcribe_url(source, options)
|
||||
else:
|
||||
with open(url, "rb") as audio:
|
||||
source = {"buffer": audio}
|
||||
response = self.client.listen.prerecorded.v("1").transcribe_file(source, options)
|
||||
content = response["results"]["channels"][0]["alternatives"][0]["transcript"]
|
||||
|
||||
doc_id = hashlib.sha256((content + url).encode()).hexdigest()
|
||||
metadata = {"url": url}
|
||||
|
||||
return {
|
||||
"doc_id": doc_id,
|
||||
"data": [
|
||||
{
|
||||
"content": content,
|
||||
"meta_data": metadata,
|
||||
}
|
||||
],
|
||||
}
|
||||
@@ -27,7 +27,7 @@ class ImageLoader(BaseLoader):
|
||||
|
||||
def _create_completion_request(self, content: str):
|
||||
return self.client.chat.completions.create(
|
||||
model="gpt-4-vision-preview", messages=[{"role": "user", "content": content}], max_tokens=self.max_tokens
|
||||
model="gpt-4o", messages=[{"role": "user", "content": content}], max_tokens=self.max_tokens
|
||||
)
|
||||
|
||||
def _process_url(self, url: str):
|
||||
|
||||
@@ -11,7 +11,8 @@ class UnstructuredLoader(BaseLoader):
|
||||
"""Load data from an Unstructured file."""
|
||||
try:
|
||||
import unstructured # noqa: F401
|
||||
from langchain_community.document_loaders import UnstructuredFileLoader
|
||||
from langchain_community.document_loaders import \
|
||||
UnstructuredFileLoader
|
||||
except ImportError:
|
||||
raise ImportError(
|
||||
'Unstructured file requires extra dependencies. Install with `pip install "unstructured[local-inference, all-docs]"`' # noqa: E501
|
||||
|
||||
@@ -8,6 +8,7 @@ except ImportError:
|
||||
raise ImportError('YouTube video requires extra dependencies. Install with `pip install youtube-transcript-api "`')
|
||||
try:
|
||||
from langchain_community.document_loaders import YoutubeLoader
|
||||
from langchain_community.document_loaders.youtube import _parse_video_id
|
||||
except ImportError:
|
||||
raise ImportError(
|
||||
'YouTube video requires extra dependencies. Install with `pip install --upgrade "embedchain[dataloaders]"`'
|
||||
@@ -21,7 +22,20 @@ from embedchain.utils.misc import clean_string
|
||||
class YoutubeVideoLoader(BaseLoader):
|
||||
def load_data(self, url):
|
||||
"""Load data from a Youtube video."""
|
||||
loader = YoutubeLoader.from_youtube_url(url, add_video_info=True)
|
||||
video_id = _parse_video_id(url)
|
||||
|
||||
languages = ["en"]
|
||||
try:
|
||||
# Fetching transcript data
|
||||
languages = [transcript.language_code for transcript in YouTubeTranscriptApi.list_transcripts(video_id)]
|
||||
transcript = YouTubeTranscriptApi.get_transcript(video_id, languages=languages)
|
||||
# convert transcript to json to avoid unicode symboles
|
||||
transcript = json.dumps(transcript, ensure_ascii=True)
|
||||
except Exception:
|
||||
logging.exception(f"Failed to fetch transcript for video {url}")
|
||||
transcript = "Unavailable"
|
||||
|
||||
loader = YoutubeLoader.from_youtube_url(url, add_video_info=True, language=languages)
|
||||
doc = loader.load()
|
||||
output = []
|
||||
if not len(doc):
|
||||
@@ -30,16 +44,7 @@ class YoutubeVideoLoader(BaseLoader):
|
||||
content = clean_string(content)
|
||||
metadata = doc[0].metadata
|
||||
metadata["url"] = url
|
||||
|
||||
video_id = url.split("v=")[1].split("&")[0]
|
||||
try:
|
||||
# Fetching transcript data
|
||||
transcript = YouTubeTranscriptApi.get_transcript(video_id, languages=["en"])
|
||||
# convert transcript to json to avoid unicode symboles
|
||||
metadata["transcript"] = json.dumps(transcript, ensure_ascii=True)
|
||||
except Exception:
|
||||
logging.exception(f"Failed to fetch transcript for video {url}")
|
||||
metadata["transcript"] = "Unavailable"
|
||||
metadata["transcript"] = transcript
|
||||
|
||||
output.append(
|
||||
{
|
||||
|
||||
@@ -41,6 +41,7 @@ class IndirectDataType(Enum):
|
||||
DROPBOX = "dropbox"
|
||||
TEXT_FILE = "text_file"
|
||||
EXCEL_FILE = "excel_file"
|
||||
AUDIO = "audio"
|
||||
|
||||
|
||||
class SpecialDataType(Enum):
|
||||
@@ -81,3 +82,4 @@ class DataType(Enum):
|
||||
DROPBOX = IndirectDataType.DROPBOX.value
|
||||
TEXT_FILE = IndirectDataType.TEXT_FILE.value
|
||||
EXCEL_FILE = IndirectDataType.EXCEL_FILE.value
|
||||
AUDIO = IndirectDataType.AUDIO.value
|
||||
|
||||
@@ -193,12 +193,15 @@ def read_env_file(env_file_path):
|
||||
dict: Dictionary of environment variables.
|
||||
"""
|
||||
env_vars = {}
|
||||
pattern = re.compile(r"(\w+)=(.*)") # compile regular expression for better performance
|
||||
with open(env_file_path, "r") as file:
|
||||
for line in file:
|
||||
lines = file.readlines() # readlines is faster as it reads all at once
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
# Ignore comments and empty lines
|
||||
if line.strip() and not line.strip().startswith("#"):
|
||||
if line and not line.startswith("#"):
|
||||
# Assume each line is in the format KEY=VALUE
|
||||
key_value_match = re.match(r"(\w+)=(.*)", line.strip())
|
||||
key_value_match = pattern.match(line)
|
||||
if key_value_match:
|
||||
key, value = key_value_match.groups()
|
||||
env_vars[key] = value
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import datetime
|
||||
import itertools
|
||||
import json
|
||||
import logging
|
||||
@@ -237,6 +238,12 @@ def detect_datatype(source: Any) -> DataType:
|
||||
logger.debug(f"Source of `{formatted_source}` detected as `docx`.")
|
||||
return DataType.DOCX
|
||||
|
||||
if url.path.endswith(
|
||||
(".mp3", ".mp4", ".mp2", ".aac", ".wav", ".flac", ".pcm", ".m4a", ".ogg", ".opus", ".webm")
|
||||
):
|
||||
logger.debug(f"Source of `{formatted_source}` detected as `audio`.")
|
||||
return DataType.AUDIO
|
||||
|
||||
if url.path.endswith(".yaml"):
|
||||
try:
|
||||
response = requests.get(source)
|
||||
@@ -407,6 +414,7 @@ def validate_config(config_data):
|
||||
"google",
|
||||
"aws_bedrock",
|
||||
"mistralai",
|
||||
"clarifai",
|
||||
"vllm",
|
||||
"groq",
|
||||
"nvidia",
|
||||
@@ -433,11 +441,14 @@ def validate_config(config_data):
|
||||
Optional("local"): bool,
|
||||
Optional("base_url"): str,
|
||||
Optional("default_headers"): dict,
|
||||
Optional("api_version"): Or(str, datetime.date),
|
||||
Optional("http_client_proxies"): Or(str, dict),
|
||||
Optional("http_async_client_proxies"): Or(str, dict),
|
||||
},
|
||||
},
|
||||
Optional("vectordb"): {
|
||||
Optional("provider"): Or(
|
||||
"chroma", "elasticsearch", "opensearch", "pinecone", "qdrant", "weaviate", "zilliz"
|
||||
"chroma", "elasticsearch", "opensearch", "lancedb", "pinecone", "qdrant", "weaviate", "zilliz"
|
||||
),
|
||||
Optional("config"): object, # TODO: add particular config schema for each provider
|
||||
},
|
||||
@@ -450,6 +461,7 @@ def validate_config(config_data):
|
||||
"azure_openai",
|
||||
"google",
|
||||
"mistralai",
|
||||
"clarifai",
|
||||
"nvidia",
|
||||
"ollama",
|
||||
"cohere",
|
||||
@@ -463,6 +475,8 @@ def validate_config(config_data):
|
||||
Optional("task_type"): str,
|
||||
Optional("vector_dimension"): int,
|
||||
Optional("base_url"): str,
|
||||
Optional("endpoint"): str,
|
||||
Optional("model_kwargs"): dict,
|
||||
},
|
||||
},
|
||||
Optional("embedding_model"): {
|
||||
@@ -474,6 +488,7 @@ def validate_config(config_data):
|
||||
"azure_openai",
|
||||
"google",
|
||||
"mistralai",
|
||||
"clarifai",
|
||||
"nvidia",
|
||||
"ollama",
|
||||
),
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user