Compare commits

...

13 Commits

Author SHA1 Message Date
Deshraj Yadav 6ced756a6b [misc] add json extra in pyproject.toml file (#886) 2023-10-31 23:55:25 -07:00
Deshraj Yadav b17268db50 [misc] remove jq as a dependency (#885) 2023-10-31 21:02:19 -07:00
Deshraj Yadav 476da37009 [version] bump package version to v0.0.87 (#883) 2023-10-31 12:40:02 -07:00
Deshraj Yadav 455f059c6f [Telemetry] Update anonymous telemetry API key (#882) 2023-10-31 12:34:54 -07:00
Deven Patel 5255a37c93 Embedchain json url support (#878)
Co-authored-by: Deven Patel <deven298@yahoo.com>
2023-10-30 16:19:11 -07:00
Deven Patel 68dc274f72 Embedchain json loader update (#876)
Co-authored-by: Deven Patel <deven298@yahoo.com>
2023-10-30 15:30:49 -07:00
Deshraj Yadav 30228f7f8e [version] Update langchain to v0.0.303 (#875) 2023-10-30 15:02:38 -07:00
Sidharth Mohanty e15ef79ca9 Lazy load loaders and chunkers (#872) 2023-10-30 11:20:38 -07:00
Deshraj Yadav bc012a7518 [Docs] Update embedchain docs and analytics (#871) 2023-10-30 00:38:35 -07:00
Sidharth Mohanty 3b4409cfad Update notebooks to work with the latest version (#870) 2023-10-29 23:06:43 -07:00
Deshraj Yadav d3726134b2 [Docs] Update docs and minor improvements in search API (#869) 2023-10-29 16:50:14 -07:00
anujshandillya 5acb7f1c55 [fix]: updated twitter logo to new X (#868) 2023-10-29 14:37:32 -07:00
Deshraj Yadav 81336668b3 [Feature]: Add posthog anonymous telemetry and update docs (#867) 2023-10-29 01:20:21 -07:00
59 changed files with 1040 additions and 734 deletions
+12 -1
View File
@@ -64,7 +64,7 @@ For example, you can use Embedchain to create an Elon Musk bot using the followi
```python
import os
from embedchain import App
from embedchain import Pipeline as App
# Create a bot instance
os.environ["OPENAI_API_KEY"] = "YOUR API KEY"
@@ -78,6 +78,17 @@ elon_bot.add("https://www.youtube.com/watch?v=RcYjXbSJBN8")
# Query the bot
elon_bot.query("How many companies does Elon Musk run and name those?")
# Answer: Elon Musk currently runs several companies. As of my knowledge, he is the CEO and lead designer of SpaceX, the CEO and product architect of Tesla, Inc., the CEO and founder of Neuralink, and the CEO and founder of The Boring Company. However, please note that this information may change over time, so it's always good to verify the latest updates.
# (Optional): Deploy app to Embedchain Platform
app.deploy()
# 🔑 Enter your Embedchain API key. You can find the API key at https://app.embedchain.ai/settings/keys/
# ec-xxxxxx
# 🛠️ Creating pipeline on the platform...
# 🎉🎉🎉 Pipeline created successfully! View your pipeline: https://app.embedchain.ai/pipelines/xxxxx
# 🛠️ Adding data to your pipeline...
# ✅ Data of type: web_page, value: https://www.forbes.com/profile/elon-musk added successfully.
```
## Examples
+5 -5
View File
@@ -24,7 +24,7 @@ Once you have obtained the key, you can use it like this:
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ['OPENAI_API_KEY'] = 'xxx'
@@ -52,7 +52,7 @@ To use Azure OpenAI embedding model, you have to set some of the azure openai re
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["OPENAI_API_TYPE"] = "azure"
os.environ["OPENAI_API_BASE"] = "https://xxx.openai.azure.com/"
@@ -90,7 +90,7 @@ GPT4All supports generating high quality embeddings of arbitrary length document
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load embedding model configuration from config.yaml file
app = App.from_config(yaml_path="config.yaml")
@@ -119,7 +119,7 @@ Hugging Face supports generating embeddings of arbitrary length documents of tex
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load embedding model configuration from config.yaml file
app = App.from_config(yaml_path="config.yaml")
@@ -150,7 +150,7 @@ Embedchain supports Google's VertexAI embeddings model through a simple interfac
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load embedding model configuration from config.yaml file
app = App.from_config(yaml_path="config.yaml")
+10 -10
View File
@@ -26,7 +26,7 @@ Once you have obtained the key, you can use it like this:
```python
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ['OPENAI_API_KEY'] = 'xxx'
@@ -41,7 +41,7 @@ If you are looking to configure the different parameters of the LLM, you can do
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ['OPENAI_API_KEY'] = 'xxx'
@@ -71,7 +71,7 @@ To use Azure OpenAI model, you have to set some of the azure openai related envi
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["OPENAI_API_TYPE"] = "azure"
os.environ["OPENAI_API_BASE"] = "https://xxx.openai.azure.com/"
@@ -110,7 +110,7 @@ To use anthropic's model, please set the `ANTHROPIC_API_KEY` which you find on t
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["ANTHROPIC_API_KEY"] = "xxx"
@@ -147,7 +147,7 @@ Once you have the API key, you are all set to use it with Embedchain.
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["COHERE_API_KEY"] = "xxx"
@@ -180,7 +180,7 @@ GPT4all is a free-to-use, locally running, privacy-aware chatbot. No GPU or inte
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load llm configuration from config.yaml file
app = App.from_config(yaml_path="config.yaml")
@@ -212,7 +212,7 @@ Once you have the key, load the app using the config yaml file:
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["JINACHAT_API_KEY"] = "xxx"
# load llm configuration from config.yaml file
@@ -248,7 +248,7 @@ Once you have the token, load the app using the config yaml file:
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["HUGGINGFACE_ACCESS_TOKEN"] = "xxx"
@@ -278,7 +278,7 @@ Once you have the token, load the app using the config yaml file:
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["REPLICATE_API_TOKEN"] = "xxx"
@@ -305,7 +305,7 @@ Setup Google Cloud Platform application credentials by following the instruction
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load llm configuration from config.yaml file
app = App.from_config(yaml_path="config.yaml")
+7 -7
View File
@@ -22,7 +22,7 @@ Utilizing a vector database alongside Embedchain is a seamless process. All you
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load chroma configuration from yaml file
app = App.from_config(yaml_path="config1.yaml")
@@ -61,7 +61,7 @@ pip install --upgrade 'embedchain[elasticsearch]'
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load elasticsearch configuration from yaml file
app = App.from_config(yaml_path="config.yaml")
@@ -89,7 +89,7 @@ pip install --upgrade 'embedchain[opensearch]'
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load opensearch configuration from yaml file
app = App.from_config(yaml_path="config.yaml")
@@ -125,7 +125,7 @@ Set the Zilliz environment variables `ZILLIZ_CLOUD_URI` and `ZILLIZ_CLOUD_TOKEN`
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ['ZILLIZ_CLOUD_URI'] = 'https://xxx.zillizcloud.com'
os.environ['ZILLIZ_CLOUD_TOKEN'] = 'xxx'
@@ -164,7 +164,7 @@ In order to use Pinecone as vector database, set the environment variables `PINE
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load pinecone configuration from yaml file
app = App.from_config(yaml_path="config.yaml")
@@ -187,7 +187,7 @@ In order to use Qdrant as a vector database, set the environment variables `QDRA
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load qdrant configuration from yaml file
app = App.from_config(yaml_path="config.yaml")
@@ -207,7 +207,7 @@ In order to use Weaviate as a vector database, set the environment variables `WE
<CodeGroup>
```python main.py
from embedchain import App
from embedchain import Pipeline as App
# load weaviate configuration from yaml file
app = App.from_config(yaml_path="config.yaml")
+1 -1
View File
@@ -5,7 +5,7 @@ title: '📊 CSV'
To add any csv file, use the data_type as `csv`. `csv` allows remote urls and conventional file paths. Headers are included for each line, so if you have an `age` column, `18` will be added as `age: 18`. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
app.add('https://people.sc.fsu.edu/~jburkardt/data/csv/airtravel.csv', data_type="csv")
+2 -2
View File
@@ -35,7 +35,7 @@ Default behavior is to create a persistent vector db in the directory **./db**.
Create a local index:
```python
from embedchain import App
from embedchain import Pipeline as App
naval_chat_bot = App()
naval_chat_bot.add("https://www.youtube.com/watch?v=3qHkcs3kG44")
@@ -45,7 +45,7 @@ naval_chat_bot.add("https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Alma
You can reuse the local index with the same code, but without adding new documents:
```python
from embedchain import App
from embedchain import Pipeline as App
naval_chat_bot = App()
print(naval_chat_bot.query("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?"))
+1 -1
View File
@@ -5,7 +5,7 @@ title: '📚🌐 Code documentation'
To add any code documentation website as a loader, use the data_type as `docs_site`. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
app.add("https://docs.embedchain.ai/", data_type="docs_site")
+1 -1
View File
@@ -7,7 +7,7 @@ title: '📄 Docx file'
To add any doc/docx file, use the data_type as `docx`. `docx` allows remote urls and conventional file paths. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
app.add('https://example.com/content/intro.docx', data_type="docx")
+1 -1
View File
@@ -5,7 +5,7 @@ title: '📝 Mdx file'
To add any `.mdx` file to your app, use the data_type (first argument to `.add()` method) as `mdx`. Note that this supports support mdx file present on machine, so this should be a file path. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
app.add('path/to/file.mdx', data_type='mdx')
+1 -1
View File
@@ -8,7 +8,7 @@ To load a notion page, use the data_type as `notion`. Since it is hard to automa
The next argument must **end** with the `notion page id`. The id is a 32-character string. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
+1 -1
View File
@@ -5,7 +5,7 @@ title: '📰 PDF file'
To add any pdf file, use the data_type as `pdf_file`. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
+1 -1
View File
@@ -5,7 +5,7 @@ title: '❓💬 Queston and answer pair'
QnA pair is a local data type. To supply your own QnA pair, use the data_type as `qna_pair` and enter a tuple. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
+1 -1
View File
@@ -5,7 +5,7 @@ title: '🗺️ Sitemap'
Add all web pages from an xml-sitemap. Filters non-text files. Use the data_type as `sitemap`. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
+1 -1
View File
@@ -7,7 +7,7 @@ title: '📝 Text'
Text is a local data type. To supply your own text, use the data_type as `text` and enter a string. The text is not processed, this can be very versatile. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
+1 -1
View File
@@ -5,7 +5,7 @@ title: '🌐📄 Web page'
To add any web page, use the data_type as `web_page`. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
+1 -1
View File
@@ -7,7 +7,7 @@ title: '🧾 XML file'
To add any xml file, use the data_type as `xml`. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
+1 -1
View File
@@ -6,7 +6,7 @@ title: '🎥📺 Youtube video'
To add any youtube video to your app, use the data_type (first argument to `.add()` method) as `youtube_video`. Eg:
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
app.add('a_valid_youtube_url_here', data_type='youtube_video')
BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 70 KiB

After

Width:  |  Height:  |  Size: 5.0 KiB

+2 -2
View File
@@ -9,7 +9,7 @@ description: 'Collections of all the frequently asked questions'
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ['OPENAI_API_KEY'] = 'xxx'
@@ -36,7 +36,7 @@ llm:
```python main.py
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ['OPENAI_API_KEY'] = 'xxx'
+92 -16
View File
@@ -3,30 +3,106 @@ title: 📚 Introduction
description: '📝 Embedchain is a Data Platform for LLMs - load, index, retrieve, and sync any unstructured data'
---
## 🤔 What is Embedchain?
## 🌐 What is Embedchain?
Embedchain abstracts the entire process of loading data, chunking it, creating embeddings, and storing it in a vector database.
Embedchain simplifies data handling by automatically processing unstructured data, breaking it into chunks, generating embeddings, and storing it in a vector database.
You can add data from different data sources using the `.add()` method. Then, simply use the `.query()` method to find answers from the added datasets.
Through various APIs, you can obtain contextual information for queries, find answers to specific questions, and engage in chat conversations using your data.
## 🔍 Search
If you want to create a Naval Ravikant bot with a YouTube video, a book in PDF format, two blog posts, and a question and answer pair, all you need to do is add the respective links. Embedchain will take care of the rest, creating a bot for you.
Embedchain lets you get most relevant context by doing semantic search over your data sources for a provided query. See the example below:
```python
from embedchain import App
from embedchain import Pipeline as App
naval_bot = App()
# Add online data
naval_bot.add("https://www.youtube.com/watch?v=3qHkcs3kG44")
naval_bot.add("https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf")
naval_bot.add("https://nav.al/feedback")
naval_bot.add("https://nav.al/agi")
naval_bot.add("The Meanings of Life", 'text', metadata={'chapter': 'philosphy'})
# Initialize app
app = App()
# Add local resources
naval_bot.add(("Who is Naval Ravikant?", "Naval Ravikant is an Indian-American entrepreneur and investor."))
# Add data source
app.add("https://www.forbes.com/profile/elon-musk")
naval_bot.query("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?")
# Answer: Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.
# Get relevant context using semantic search
context = app.search("What is the net worth of Elon?", num_documents=2)
print(context)
# Context:
# [
# {
# 'context': 'Elon Musk PROFILEElon MuskCEO, Tesla$221.9BReal Time Net Worthas of 10/29/23Reflects change since 5 pm ET of prior trading day. 1 in the world todayPhoto by Martin Schoeller for ForbesAbout Elon MuskElon Musk cofounded six companies, including electric car maker Tesla, rocket producer SpaceX and tunneling startup Boring Company.He owns about 21% of Tesla between stock and options, but has pledged more than half his shares as collateral for personal loans of up to $3.5 billion.SpaceX, founded in',
# 'source': 'https://www.forbes.com/profile/elon-musk',
# 'document_id': 'some_document_id'
# },
# {
# 'context': 'company, which is now called X.Wealth HistoryHOVER TO REVEAL NET WORTH BY YEARForbes Lists 1Forbes 400 (2023)The Richest Person In Every State (2023) 2Billionaires (2023) 1Innovative Leaders (2019) 25Powerful People (2018) 12Richest In Tech (2017)Global Game Changers (2016)More ListsPersonal StatsAge52Source of WealthTesla, SpaceX, Self MadeSelf-Made Score8Philanthropy Score1ResidenceAustin, TexasCitizenshipUnited StatesMarital StatusSingleChildren11EducationBachelor of Arts/Science, University',
# 'source': 'https://www.forbes.com/profile/elon-musk',
# 'document_id': 'some_document_id'
# }
# ]
```
## ❓Query
Embedchain empowers developers to ask questions and receive relevant answers through a user-friendly query API. Refer to the following example to learn how to utilize the query API:
```python
from embedchain import Pipeline as App
# Initialize app
app = App()
# Add data source
app.add("https://www.forbes.com/profile/elon-musk")
# Get relevant answer for your query
answer = app.query("What is the net worth of Elon?")
print(answer)
# Answer: The net worth of Elon Musk is $221.9 billion.
```
## 💬 Chat
Embedchain allows easy chatting over your data sources using a user-friendly chat API. Check out the example below to understand how to use the chat API:
```python
from embedchain import Pipeline as App
# Initialize app
app = App()
# Add data source
app.add("https://www.forbes.com/profile/elon-musk")
# Chat on your data using `.chat()`
answer = app.chat("How much did Elon pay for Twitter?")
print(answer)
# Answer: Elon Musk paid $44 billion for Twitter.
```
## 🚀 Deploy
Embedchain enables developers to deploy their LLM-powered apps in production using the Embedchain platform. The platform offers free access to context on your data through its REST API. Once the pipeline is deployed, you can update your data sources anytime after deployment.
See the example below on how to use the deploy API:
```python
from embedchain import Pipeline as App
# Initialize app
app = App()
# Add data source
app.add("https://www.forbes.com/profile/elon-musk")
# Deploy your pipeline to Embedchain Platform
app.deploy()
# 🔑 Enter your Embedchain API key. You can find the API key at https://app.embedchain.ai/settings/keys/
# ec-xxxxxx
# 🛠️ Creating pipeline on the platform...
# 🎉🎉🎉 Pipeline created successfully! View your pipeline: https://app.embedchain.ai/pipelines/xxxxx
# 🛠️ Adding data to your pipeline...
# ✅ Data of type: web_page, value: https://www.forbes.com/profile/elon-musk added successfully.
```
## 🚀 How it works?
+35 -12
View File
@@ -16,23 +16,36 @@ Creating an app involves 3 steps:
<Steps>
<Step title="⚙️ Import app instance">
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
```
</Step>
<Step title="🗃️ Add data sources">
```python
# Add different data sources
elon_bot.add("https://en.wikipedia.org/wiki/Elon_Musk")
elon_bot.add("https://www.forbes.com/profile/elon-musk")
app.add("https://en.wikipedia.org/wiki/Elon_Musk")
app.add("https://www.forbes.com/profile/elon-musk")
# You can also add local data sources such as pdf, csv files etc.
# elon_bot.add("/path/to/file.pdf")
# app.add("/path/to/file.pdf")
```
</Step>
<Step title="💬 Query or chat on your data and get answers">
<Step title="💬 Query or chat or search context on your data">
```python
elon_bot.query("What is the net worth of Elon Musk today?")
app.query("What is the net worth of Elon Musk today?")
# Answer: The net worth of Elon Musk today is $258.7 billion.
```
</Step>
<Step title="🚀 (Optional) Deploy your pipeline to Embedchain Platform">
```python
app.deploy()
# 🔑 Enter your Embedchain API key. You can find the API key at https://app.embedchain.ai/settings/keys/
# ec-xxxxxx
# 🛠️ Creating pipeline on the platform...
# 🎉🎉🎉 Pipeline created successfully! View your pipeline: https://app.embedchain.ai/pipelines/xxxxx
# 🛠️ Adding data to your pipeline...
# ✅ Data of type: web_page, value: https://www.forbes.com/profile/elon-musk added successfully.
```
</Step>
</Steps>
@@ -41,18 +54,28 @@ Putting it together, you can run your first app using the following code. Make s
```python
import os
from embedchain import App
from embedchain import Pipeline as App
os.environ["OPENAI_API_KEY"] = "xxx"
elon_bot = App()
app = App()
# Add different data sources
elon_bot.add("https://en.wikipedia.org/wiki/Elon_Musk")
elon_bot.add("https://www.forbes.com/profile/elon-musk")
app.add("https://en.wikipedia.org/wiki/Elon_Musk")
app.add("https://www.forbes.com/profile/elon-musk")
# You can also add local data sources such as pdf, csv files etc.
# elon_bot.add("/path/to/file.pdf")
# app.add("/path/to/file.pdf")
response = elon_bot.query("What is the net worth of Elon Musk today?")
response = app.query("What is the net worth of Elon Musk today?")
print(response)
# Answer: The net worth of Elon Musk today is $258.7 billion.
app.deploy()
# 🔑 Enter your Embedchain API key. You can find the API key at https://app.embedchain.ai/settings/keys/
# ec-xxxxxx
# 🛠️ Creating pipeline on the platform...
# 🎉🎉🎉 Pipeline created successfully! View your pipeline: https://app.embedchain.ai/pipelines/xxxxx
# 🛠️ Adding data to your pipeline...
# ✅ Data of type: web_page, value: https://www.forbes.com/profile/elon-musk added successfully.
```
Binary file not shown.

Before

Width:  |  Height:  |  Size: 256 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 83 KiB

+1 -1
View File
@@ -39,7 +39,7 @@ os.environ['LANGCHAIN_PROJECT] = <your-project>
```python
from embedchain import App
from embedchain import Pipeline as App
app = App()
app.add("https://en.wikipedia.org/wiki/Elon_Musk")
+23 -14
View File
File diff suppressed because one or more lines are too long

Before

Width:  |  Height:  |  Size: 42 KiB

After

Width:  |  Height:  |  Size: 3.1 KiB

+23 -14
View File
File diff suppressed because one or more lines are too long

Before

Width:  |  Height:  |  Size: 42 KiB

After

Width:  |  Height:  |  Size: 3.1 KiB

+33 -12
View File
@@ -7,9 +7,20 @@
},
"favicon": "/favicon.png",
"colors": {
"primary": "#12A7D3",
"light": "#81D7F7",
"dark": "#004E7A"
"primary": "#2B48EE",
"light": "#2B48EE",
"dark": "#2B48EE",
"background": {
"dark": "#020415"
}
},
"metadata": {
"og:image": "/images/og.png",
"twitter:site": "@embedchain"
},
"topAnchor": {
"name": "Documentation",
"icon": "book-open"
},
"topbarLinks": [
{
@@ -26,8 +37,11 @@
}
],
"topbarCtaButton": {
"name": "GitHub",
"url": "https://embedchain.ai"
"name": "Get started",
"url": "https://app.embedchain.ai"
},
"primaryTab": {
"name": "Docs"
},
"navigation": [
{
@@ -71,10 +85,6 @@
"group": "Examples",
"pages": ["examples/full_stack", "examples/api_server", "examples/discord_bot", "examples/slack_bot", "examples/telegram_bot", "examples/whatsapp_bot", "examples/poe_bot"]
},
{
"group": "Pipelines",
"pages": ["pipelines/quickstart"]
},
{
"group": "Community",
"pages": [
@@ -104,7 +114,6 @@
}
],
"footerSocials": {
"website": "https://embedchain.ai",
"github": "https://github.com/embedchain/embedchain",
@@ -113,7 +122,19 @@
"twitter": "https://twitter.com/embedchain",
"linkedin": "https://www.linkedin.com/company/embedchain"
},
"backgroundImage": "/background.png",
"isWhiteLabeled": true,
"feedback.thumbsRating": true
"analytics": {
"posthog": {
"apiKey": "phc_PHQDA5KwztijnSojsxJ2c1DuJd52QCzJzT2xnSGvjN2",
"apiHost": "https://app.embedchain.ai/ingest"
}
},
"feedback": {
"suggestEdit": true,
"raiseIssue": true,
"thumbsRating": true
},
"search": {
"prompt": "✨ Search embedchain docs..."
}
}
-44
View File
@@ -1,44 +0,0 @@
---
title: '🚀 Pipelines'
description: '💡 Start building LLM powered data pipelines in 1 minute'
---
Embedchain lets you build data pipelines on your own data sources and deploy it in production in less than a minute. It can load, index, retrieve, and sync any unstructured data.
Install embedchain python package:
```bash
pip install embedchain
```
Creating a pipeline involves 3 steps:
<Steps>
<Step title="⚙️ Import pipeline instance">
```python
from embedchain import Pipeline
p = Pipeline(name="Elon Musk")
```
</Step>
<Step title="🗃️ Add data sources">
```python
# Add different data sources
p.add("https://en.wikipedia.org/wiki/Elon_Musk")
p.add("https://www.forbes.com/profile/elon-musk")
# You can also add local data sources such as pdf, csv files etc.
# p.add("/path/to/file.pdf")
```
</Step>
<Step title="💬 Deploy your pipeline to Embedchain platform">
```python
p.deploy()
```
</Step>
</Steps>
That's it. Now, head to the [Embedchain platform](https://app.embedchain.ai) and your pipeline is available there. Make sure to set the `OPENAI_API_KEY` 🔑 environment variable in the code.
After you deploy your pipeline to Embedchain platform, you can still add more data sources and update the pipeline multiple times.
Here is a Google Colab notebook for you to get started: [![Open in Colab](https://camo.githubusercontent.com/84f0493939e0c4de4e6dbe113251b4bfb5353e57134ffd9fcab6b8714514d4d1/68747470733a2f2f636f6c61622e72657365617263682e676f6f676c652e636f6d2f6173736574732f636f6c61622d62616467652e737667)](https://colab.research.google.com/drive/1YVXaBO4yqlHZY4ho67GCJ6aD4CHNiScD?usp=sharing)
+1 -1
View File
@@ -15,7 +15,7 @@ class AppConfig(BaseAppConfig):
self,
log_level: str = "WARNING",
id: Optional[str] = None,
collect_metrics: Optional[bool] = None,
collect_metrics: Optional[bool] = True,
collection_name: Optional[str] = None,
):
"""
+1 -1
View File
@@ -16,7 +16,7 @@ class PipelineConfig(BaseAppConfig):
log_level: str = "WARNING",
id: Optional[str] = None,
name: Optional[str] = None,
collect_metrics: Optional[bool] = False,
collect_metrics: Optional[bool] = True,
):
"""
Initializes a configuration class instance for an App. This is the simplest form of an embedchain app.
+47 -89
View File
@@ -1,41 +1,10 @@
from importlib import import_module
from embedchain.chunkers.base_chunker import BaseChunker
from embedchain.chunkers.docs_site import DocsSiteChunker
from embedchain.chunkers.docx_file import DocxFileChunker
from embedchain.chunkers.gmail import GmailChunker
from embedchain.chunkers.images import ImagesChunker
from embedchain.chunkers.json import JSONChunker
from embedchain.chunkers.mdx import MdxChunker
from embedchain.chunkers.notion import NotionChunker
from embedchain.chunkers.openapi import OpenAPIChunker
from embedchain.chunkers.pdf_file import PdfFileChunker
from embedchain.chunkers.qna_pair import QnaPairChunker
from embedchain.chunkers.sitemap import SitemapChunker
from embedchain.chunkers.table import TableChunker
from embedchain.chunkers.text import TextChunker
from embedchain.chunkers.unstructured_file import UnstructuredFileChunker
from embedchain.chunkers.web_page import WebPageChunker
from embedchain.chunkers.xml import XmlChunker
from embedchain.chunkers.youtube_video import YoutubeVideoChunker
from embedchain.config import AddConfig
from embedchain.config.add_config import ChunkerConfig, LoaderConfig
from embedchain.helper.json_serializable import JSONSerializable
from embedchain.loaders.base_loader import BaseLoader
from embedchain.loaders.csv import CsvLoader
from embedchain.loaders.docs_site_loader import DocsSiteLoader
from embedchain.loaders.docx_file import DocxFileLoader
from embedchain.loaders.gmail import GmailLoader
from embedchain.loaders.images import ImagesLoader
from embedchain.loaders.json import JSONLoader
from embedchain.loaders.local_qna_pair import LocalQnaPairLoader
from embedchain.loaders.local_text import LocalTextLoader
from embedchain.loaders.mdx import MdxLoader
from embedchain.loaders.openapi import OpenAPILoader
from embedchain.loaders.pdf_file import PdfFileLoader
from embedchain.loaders.sitemap import SitemapLoader
from embedchain.loaders.unstructured_file import UnstructuredLoader
from embedchain.loaders.web_page import WebPageLoader
from embedchain.loaders.xml import XmlLoader
from embedchain.loaders.youtube_video import YoutubeVideoLoader
from embedchain.models.data_type import DataType
@@ -58,6 +27,11 @@ class DataFormatter(JSONSerializable):
self.loader = self._get_loader(data_type=data_type, config=config.loader)
self.chunker = self._get_chunker(data_type=data_type, config=config.chunker)
def _lazy_load(self, module_path: str):
module_path, class_name = module_path.rsplit(".", 1)
module = import_module(module_path)
return getattr(module, class_name)
def _get_loader(self, data_type: DataType, config: LoaderConfig) -> BaseLoader:
"""
Returns the appropriate data loader for the given data type.
@@ -71,71 +45,55 @@ class DataFormatter(JSONSerializable):
:rtype: BaseLoader
"""
loaders = {
DataType.YOUTUBE_VIDEO: YoutubeVideoLoader,
DataType.PDF_FILE: PdfFileLoader,
DataType.WEB_PAGE: WebPageLoader,
DataType.QNA_PAIR: LocalQnaPairLoader,
DataType.TEXT: LocalTextLoader,
DataType.DOCX: DocxFileLoader,
DataType.SITEMAP: SitemapLoader,
DataType.XML: XmlLoader,
DataType.DOCS_SITE: DocsSiteLoader,
DataType.CSV: CsvLoader,
DataType.MDX: MdxLoader,
DataType.IMAGES: ImagesLoader,
DataType.UNSTRUCTURED: UnstructuredLoader,
DataType.JSON: JSONLoader,
DataType.OPENAPI: OpenAPILoader,
DataType.GMAIL: GmailLoader,
DataType.YOUTUBE_VIDEO: "embedchain.loaders.youtube_video.YoutubeVideoLoader",
DataType.PDF_FILE: "embedchain.loaders.pdf_file.PdfFileLoader",
DataType.WEB_PAGE: "embedchain.loaders.web_page.WebPageLoader",
DataType.QNA_PAIR: "embedchain.loaders.local_qna_pair.LocalQnaPairLoader",
DataType.TEXT: "embedchain.loaders.local_text.LocalTextLoader",
DataType.DOCX: "embedchain.loaders.docx_file.DocxFileLoader",
DataType.SITEMAP: "embedchain.loaders.sitemap.SitemapLoader",
DataType.XML: "embedchain.loaders.xml.XmlLoader",
DataType.DOCS_SITE: "embedchain.loaders.docs_site_loader.DocsSiteLoader",
DataType.CSV: "embedchain.loaders.csv.CsvLoader",
DataType.MDX: "embedchain.loaders.mdx.MdxLoader",
DataType.IMAGES: "embedchain.loaders.images.ImagesLoader",
DataType.UNSTRUCTURED: "embedchain.loaders.unstructured_file.UnstructuredLoader",
DataType.JSON: "embedchain.loaders.json.JSONLoader",
DataType.OPENAPI: "embedchain.loaders.openapi.OpenAPILoader",
DataType.GMAIL: "embedchain.loaders.gmail.GmailLoader",
DataType.NOTION: "embedchain.loaders.notion.NotionLoader",
}
lazy_loaders = {DataType.NOTION}
if data_type in loaders:
loader_class: type = loaders[data_type]
loader: BaseLoader = loader_class()
return loader
elif data_type in lazy_loaders:
if data_type == DataType.NOTION:
from embedchain.loaders.notion import NotionLoader
return NotionLoader()
else:
raise ValueError(f"Unsupported data type: {data_type}")
loader_class: type = self._lazy_load(loaders[data_type])
return loader_class()
else:
raise ValueError(f"Unsupported data type: {data_type}")
def _get_chunker(self, data_type: DataType, config: ChunkerConfig) -> BaseChunker:
"""Returns the appropriate chunker for the given data type.
:param data_type: The type of the data to chunk.
:type data_type: DataType
:param config: Config to initialize the chunker with.
:type config: ChunkerConfig
:raises ValueError: If an unsupported data type is provided.
:return: The chunker for the given data type.
:rtype: BaseChunker
"""
"""Returns the appropriate chunker for the given data type (updated for lazy loading)."""
chunker_classes = {
DataType.YOUTUBE_VIDEO: YoutubeVideoChunker,
DataType.PDF_FILE: PdfFileChunker,
DataType.WEB_PAGE: WebPageChunker,
DataType.QNA_PAIR: QnaPairChunker,
DataType.TEXT: TextChunker,
DataType.DOCX: DocxFileChunker,
DataType.DOCS_SITE: DocsSiteChunker,
DataType.SITEMAP: SitemapChunker,
DataType.NOTION: NotionChunker,
DataType.CSV: TableChunker,
DataType.MDX: MdxChunker,
DataType.IMAGES: ImagesChunker,
DataType.XML: XmlChunker,
DataType.UNSTRUCTURED: UnstructuredFileChunker,
DataType.JSON: JSONChunker,
DataType.OPENAPI: OpenAPIChunker,
DataType.GMAIL: GmailChunker,
DataType.YOUTUBE_VIDEO: "embedchain.chunkers.youtube_video.YoutubeVideoChunker",
DataType.PDF_FILE: "embedchain.chunkers.pdf_file.PdfFileChunker",
DataType.WEB_PAGE: "embedchain.chunkers.web_page.WebPageChunker",
DataType.QNA_PAIR: "embedchain.chunkers.qna_pair.QnaPairChunker",
DataType.TEXT: "embedchain.chunkers.text.TextChunker",
DataType.DOCX: "embedchain.chunkers.docx_file.DocxFileChunker",
DataType.SITEMAP: "embedchain.chunkers.sitemap.SitemapChunker",
DataType.XML: "embedchain.chunkers.xml.XmlChunker",
DataType.DOCS_SITE: "embedchain.chunkers.docs_site.DocsSiteChunker",
DataType.CSV: "embedchain.chunkers.table.TableChunker",
DataType.MDX: "embedchain.chunkers.mdx.MdxChunker",
DataType.IMAGES: "embedchain.chunkers.images.ImagesChunker",
DataType.UNSTRUCTURED: "embedchain.chunkers.unstructured_file.UnstructuredFileChunker",
DataType.JSON: "embedchain.chunkers.json.JSONChunker",
DataType.OPENAPI: "embedchain.chunkers.openapi.OpenAPIChunker",
DataType.GMAIL: "embedchain.chunkers.gmail.GmailChunker",
DataType.NOTION: "embedchain.chunkers.notion.NotionChunker",
}
if data_type in chunker_classes:
chunker_class: type = chunker_classes[data_type]
chunker: BaseChunker = chunker_class(config)
chunker_class = self._lazy_load(chunker_classes[data_type])
chunker = chunker_class(config)
chunker.set_data_type(data_type)
return chunker
else:
+17 -74
View File
@@ -1,18 +1,13 @@
import hashlib
import importlib.metadata
import json
import logging
import os
import sqlite3
import threading
import uuid
from pathlib import Path
from typing import Any, Dict, List, Optional
import requests
from dotenv import load_dotenv
from langchain.docstore.document import Document
from tenacity import retry, stop_after_attempt, wait_fixed
from embedchain.chunkers.base_chunker import BaseChunker
from embedchain.config import AddConfig, BaseLlmConfig
@@ -24,6 +19,7 @@ from embedchain.llm.base import BaseLlm
from embedchain.loaders.base_loader import BaseLoader
from embedchain.models.data_type import (DataType, DirectDataType,
IndirectDataType, SpecialDataType)
from embedchain.telemetry.posthog import AnonymousTelemetry
from embedchain.utils import detect_datatype
from embedchain.vectordb.base import BaseVectorDB
@@ -89,9 +85,8 @@ class EmbedChain(JSONSerializable):
self.user_asks = []
# Send anonymous telemetry
self.s_id = self.config.id if self.config.id else str(uuid.uuid4())
self.u_id = self._load_or_generate_user_id()
self._telemetry_props = {"class": self.__class__.__name__}
self.telemetry = AnonymousTelemetry(enabled=self.config.collect_metrics)
# Establish a connection to the SQLite database
self.connection = sqlite3.connect(SQLITE_PATH)
self.cursor = self.connection.cursor()
@@ -111,12 +106,8 @@ class EmbedChain(JSONSerializable):
"""
)
self.connection.commit()
# NOTE: Uncomment the next two lines when running tests to see if any test fires a telemetry event.
# if (self.config.collect_metrics):
# raise ConnectionRefusedError("Collection of metrics should not be allowed.")
thread_telemetry = threading.Thread(target=self._send_telemetry_event, args=("init",))
thread_telemetry.start()
# Send anonymous telemetry
self.telemetry.capture(event_name="init", properties=self._telemetry_props)
@property
def collect_metrics(self):
@@ -138,29 +129,6 @@ class EmbedChain(JSONSerializable):
raise ValueError(f"Boolean value expected but got {type(value)}.")
self.llm.online = value
def _load_or_generate_user_id(self) -> str:
"""
Loads the user id from the config file if it exists, otherwise generates a new
one and saves it to the config file.
:return: user id
:rtype: str
"""
if not os.path.exists(CONFIG_DIR):
os.makedirs(CONFIG_DIR)
if os.path.exists(CONFIG_FILE):
with open(CONFIG_FILE, "r") as f:
data = json.load(f)
if "user_id" in data:
return data["user_id"]
u_id = str(uuid.uuid4())
with open(CONFIG_FILE, "w") as f:
json.dump({"user_id": u_id}, f)
return u_id
def add(
self,
source: Any,
@@ -259,9 +227,14 @@ class EmbedChain(JSONSerializable):
# it's quicker to check the variable twice than to count words when they won't be submitted.
word_count = data_formatter.chunker.get_word_count(documents)
extra_metadata = {"data_type": data_type.value, "word_count": word_count, "chunks_count": new_chunks}
thread_telemetry = threading.Thread(target=self._send_telemetry_event, args=("add", extra_metadata))
thread_telemetry.start()
# Send anonymous telemetry
event_properties = {
**self._telemetry_props,
"data_type": data_type.value,
"word_count": word_count,
"chunks_count": new_chunks,
}
self.telemetry.capture(event_name="add", properties=event_properties)
return source_hash
@@ -535,9 +508,7 @@ class EmbedChain(JSONSerializable):
answer = self.llm.query(input_query=input_query, contexts=contexts, config=config, dry_run=dry_run)
# Send anonymous telemetry
thread_telemetry = threading.Thread(target=self._send_telemetry_event, args=("query",))
thread_telemetry.start()
self.telemetry.capture(event_name="query", properties=self._telemetry_props)
return answer
def chat(
@@ -569,10 +540,8 @@ class EmbedChain(JSONSerializable):
"""
contexts = self.retrieve_from_database(input_query=input_query, config=config, where=where)
answer = self.llm.chat(input_query=input_query, contexts=contexts, config=config, dry_run=dry_run)
# Send anonymous telemetry
thread_telemetry = threading.Thread(target=self._send_telemetry_event, args=("chat",))
thread_telemetry.start()
self.telemetry.capture(event_name="chat", properties=self._telemetry_props)
return answer
@@ -608,34 +577,8 @@ class EmbedChain(JSONSerializable):
Resets the database. Deletes all embeddings irreversibly.
`App` does not have to be reinitialized after using this method.
"""
# Send anonymous telemetry
thread_telemetry = threading.Thread(target=self._send_telemetry_event, args=("reset",))
thread_telemetry.start()
self.db.reset()
self.cursor.execute("DELETE FROM data_sources WHERE pipeline_id = ?", (self.config.id,))
self.connection.commit()
@retry(stop=stop_after_attempt(3), wait=wait_fixed(1))
def _send_telemetry_event(self, method: str, extra_metadata: Optional[dict] = None):
"""
Send telemetry event to the embedchain server. This is anonymous. It can be toggled off in `AppConfig`.
"""
if not self.config.collect_metrics:
return
with threading.Lock():
url = "https://api.embedchain.ai/api/v1/telemetry/"
metadata = {
"s_id": self.s_id,
"version": importlib.metadata.version(__package__ or __name__),
"method": method,
"language": "py",
"u_id": self.u_id,
}
if extra_metadata:
metadata.update(extra_metadata)
response = requests.post(url, json={"metadata": metadata})
if response.status_code != 200:
logging.warning(f"Telemetry event failed with status code {response.status_code}")
# Send anonymous telemetry
self.telemetry.capture(event_name="reset", properties=self._telemetry_props)
+40 -8
View File
@@ -1,24 +1,56 @@
import hashlib
import json
import os
import re
from langchain.document_loaders.json_loader import \
JSONLoader as LangchainJSONLoader
import requests
from embedchain.loaders.base_loader import BaseLoader
langchain_json_jq_schema = 'to_entries | map("\(.key): \(.value|tostring)") | .[]'
VALID_URL_PATTERN = "^https:\/\/[0-9A-z.]+.[0-9A-z.]+.[a-z]+\/.*\.json$"
class JSONLoader(BaseLoader):
@staticmethod
def load_data(content):
"""Load a json file. Each data point is a key value pair."""
try:
from llama_hub.jsondata.base import \
JSONDataReader as LLHBUBJSONLoader
except ImportError:
raise Exception(
f"Couldn't import the required packages to load {content}, \
Do `pip install --upgrade 'embedchain[json]`"
)
loader = LLHBUBJSONLoader()
if not isinstance(content, str):
print(f"Invaid content input. Provide the correct path to the json file saved locally in {content}")
data = []
data_content = []
loader = LangchainJSONLoader(content, text_content=False, jq_schema=langchain_json_jq_schema)
docs = loader.load()
# Load json data from various sources. TODO: add support for dictionary
if os.path.isfile(content):
with open(content, "r") as json_file:
json_data = json.load(json_file)
elif re.match(VALID_URL_PATTERN, content):
response = requests.get(content)
if response.status_code == 200:
json_data = response.json()
else:
raise ValueError(
f"Loading data from the given url: {content} failed. \
Make sure the url is working."
)
else:
raise ValueError(f"Invalid content to load json data from: {content}")
docs = loader.load_data(json_data)
for doc in docs:
meta_data = doc.metadata
data.append({"content": doc.page_content, "meta_data": {"url": content, "row": meta_data["seq_num"]}})
data_content.append(doc.page_content)
doc_content = doc.text
data.append({"content": doc_content, "meta_data": {"url": content}})
data_content.append(doc_content)
doc_id = hashlib.sha256((content + ", ".join(data_content)).encode()).hexdigest()
return {"doc_id": doc_id, "data": data}
+29 -4
View File
@@ -18,6 +18,7 @@ from embedchain.factory import EmbedderFactory, LlmFactory, VectorDBFactory
from embedchain.helper.json_serializable import register_deserializable
from embedchain.llm.base import BaseLlm
from embedchain.llm.openai import OpenAILlm
from embedchain.telemetry.posthog import AnonymousTelemetry
from embedchain.vectordb.base import BaseVectorDB
from embedchain.vectordb.chroma import ChromaDB
@@ -109,8 +110,9 @@ class Pipeline(EmbedChain):
self.llm = llm or OpenAILlm()
self._init_db()
# setup user id and directory
self.u_id = self._load_or_generate_user_id()
# Send anonymous telemetry
self._telemetry_props = {"class": self.__class__.__name__}
self.telemetry = AnonymousTelemetry(enabled=self.config.collect_metrics)
# Establish a connection to the SQLite database
self.connection = sqlite3.connect(SQLITE_PATH)
@@ -131,8 +133,10 @@ class Pipeline(EmbedChain):
"""
)
self.connection.commit()
# Send anonymous telemetry
self.telemetry.capture(event_name="init", properties=self._telemetry_props)
self.user_asks = [] # legacy defaults
self.user_asks = []
if self.auto_deploy:
self.deploy()
@@ -219,15 +223,28 @@ class Pipeline(EmbedChain):
"""
Search for similar documents related to the query in the vector database.
"""
# Send anonymous telemetry
self.telemetry.capture(event_name="search", properties=self._telemetry_props)
# TODO: Search will call the endpoint rather than fetching the data from the db itself when deploy=True.
if self.id is None:
where = {"app_id": self.local_id}
return self.db.query(
context = self.db.query(
query,
n_results=num_documents,
where=where,
skip_embedding=False,
)
result = []
for c in context:
result.append(
{
"context": c[0],
"source": c[1],
"document_id": c[2],
}
)
return result
else:
# Make API call to the backend to get the results
NotImplementedError("Search is not implemented yet for the prod mode.")
@@ -312,6 +329,9 @@ class Pipeline(EmbedChain):
data_hash, data_type, data_value = result[1], result[2], result[3]
self._process_and_upload_data(data_hash, data_type, data_value)
# Send anonymous telemetry
self.telemetry.capture(event_name="deploy", properties=self._telemetry_props)
@classmethod
def from_config(cls, yaml_path: str, auto_deploy: bool = False):
"""
@@ -347,6 +367,11 @@ class Pipeline(EmbedChain):
embedding_model = EmbedderFactory.create(
embedding_model_provider, embedding_model_config_data.get("config", {})
)
# Send anonymous telemetry
event_properties = {"init_type": "yaml_config"}
AnonymousTelemetry().capture(event_name="init", properties=event_properties)
return cls(
config=pipeline_config,
llm=llm,
View File
+67
View File
@@ -0,0 +1,67 @@
import json
import logging
import os
import uuid
from pathlib import Path
from posthog import Posthog
import embedchain
HOME_DIR = str(Path.home())
CONFIG_DIR = os.path.join(HOME_DIR, ".embedchain")
CONFIG_FILE = os.path.join(CONFIG_DIR, "config.json")
logger = logging.getLogger(__name__)
class AnonymousTelemetry:
def __init__(self, host="https://app.posthog.com", enabled=True):
self.project_api_key = "phc_PHQDA5KwztijnSojsxJ2c1DuJd52QCzJzT2xnSGvjN2"
self.host = host
self.posthog = Posthog(project_api_key=self.project_api_key, host=self.host)
self.user_id = self.get_user_id()
self.enabled = enabled
# Check if telemetry tracking is disabled via environment variable
if "EC_TELEMETRY" in os.environ and os.environ["EC_TELEMETRY"].lower() not in [
"1",
"true",
"yes",
]:
self.enabled = False
if not self.enabled:
self.posthog.disabled = True
# Silence posthog logging
posthog_logger = logging.getLogger("posthog")
posthog_logger.disabled = True
def get_user_id(self):
if not os.path.exists(CONFIG_DIR):
os.makedirs(CONFIG_DIR)
if os.path.exists(CONFIG_FILE):
with open(CONFIG_FILE, "r") as f:
data = json.load(f)
if "user_id" in data:
return data["user_id"]
user_id = str(uuid.uuid4())
with open(CONFIG_FILE, "w") as f:
json.dump({"user_id": user_id}, f)
return user_id
def capture(self, event_name, properties=None):
default_properties = {
"version": embedchain.__version__,
"language": "python",
"pid": os.getpid(),
}
properties.update(default_properties)
try:
self.posthog.capture(self.user_id, event_name, properties)
except Exception:
logger.exception(f"Failed to send telemetry {event_name=}")
+1 -1
View File
@@ -38,7 +38,7 @@ class ChromaDB(BaseVectorDB):
else:
self.config = ChromaDbConfig()
self.settings = Settings()
self.settings = Settings(anonymized_telemetry=False)
self.settings.allow_reset = self.config.allow_reset if hasattr(self.config, "allow_reset") else False
if self.config.chroma_settings:
for key, value in self.config.chroma_settings.items():
@@ -1,24 +1,10 @@
<svg
fill="currentColor"
version="1.1"
xmlns="http://www.w3.org/2000/svg"
xmlns:xlink="http://www.w3.org/1999/xlink"
viewBox="0 0 512 512"
xml:space="preserve"
>
<g id="SVGRepo_bgCarrier" stroke-width="0"></g>
<g
id="SVGRepo_tracerCarrier"
stroke-linecap="round"
stroke-linejoin="round"
></g>
<g id="SVGRepo_iconCarrier">
<g id="7935ec95c421cee6d86eb22ecd12f847">
<path
style="display: inline;"
d="M459.186,151.787c0.203,4.501,0.305,9.023,0.305,13.565 c0,138.542-105.461,298.285-298.274,298.285c-59.209,0-114.322-17.357-160.716-47.104c8.212,0.973,16.546,1.47,25.012,1.47 c49.121,0,94.318-16.759,130.209-44.884c-45.887-0.841-84.596-31.154-97.938-72.804c6.408,1.227,12.968,1.886,19.73,1.886 c9.55,0,18.816-1.287,27.617-3.68c-47.955-9.633-84.1-52.001-84.1-102.795c0-0.446,0-0.882,0.011-1.318 c14.133,7.847,30.294,12.562,47.488,13.109c-28.134-18.796-46.637-50.885-46.637-87.262c0-19.212,5.16-37.218,14.193-52.7 c51.707,63.426,128.941,105.156,216.072,109.536c-1.784-7.675-2.718-15.674-2.718-23.896c0-57.891,46.941-104.832,104.832-104.832 c30.173,0,57.404,12.734,76.525,33.102c23.887-4.694,46.313-13.423,66.569-25.438c-7.827,24.485-24.434,45.025-46.089,58.002 c21.209-2.535,41.426-8.171,60.222-16.505C497.448,118.542,479.666,137.004,459.186,151.787z"
>
</path>
</g>
</g>
</svg>
<?xml version="1.0" encoding="utf-8"?>
<!-- Generator: Adobe Illustrator 27.5.0, SVG Export Plug-In . SVG Version: 6.00 Build 0) -->
<svg version="1.1" id="svg5" xmlns:svg="http://www.w3.org/2000/svg"
xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" x="0px" y="0px" viewBox="0 0 1668.56 1221.19"
style="enable-background:new 0 0 1668.56 1221.19;" xml:space="preserve">
<g id="layer1" transform="translate(52.390088,-25.058597)">
<path id="path1009" d="M283.94,167.31l386.39,516.64L281.5,1104h87.51l340.42-367.76L984.48,1104h297.8L874.15,558.3l361.92-390.99
h-87.51l-313.51,338.7l-253.31-338.7H283.94z M412.63,231.77h136.81l604.13,807.76h-136.81L412.63,231.77z"/>
</g>
</svg>

Before

Width:  |  Height:  |  Size: 1.3 KiB

After

Width:  |  Height:  |  Size: 722 B

+10 -1
View File
@@ -34,6 +34,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -54,7 +63,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\"\n",
"os.environ[\"ANTHROPIC_API_KEY\"] = \"xxx\""
+11 -1
View File
@@ -26,6 +26,16 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "692ff37b",
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"id": "ac982a56",
@@ -44,7 +54,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_TYPE\"] = \"azure\"\n",
"os.environ[\"OPENAI_API_BASE\"] = \"https://xxx.openai.azure.com/\"\n",
+87 -78
View File
@@ -1,84 +1,84 @@
{
"nbformat": 4,
"nbformat_minor": 0,
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"name": "python3",
"display_name": "Python 3"
},
"language_info": {
"name": "python"
}
},
"cells": [
{
"cell_type": "markdown",
"source": [
"## Cookbook for using ChromaDB with Embedchain"
],
"metadata": {
"id": "b02n_zJ_hl3d"
}
},
"source": [
"## Cookbook for using ChromaDB with Embedchain"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-1: Install embedchain package"
],
"metadata": {
"id": "gyJ6ui2vhtMY"
}
},
"source": [
"### Step-1: Install embedchain package"
]
},
{
"cell_type": "code",
"source": [
"!pip install embedchain"
],
"execution_count": null,
"metadata": {
"id": "-NbXjAdlh0vJ"
},
"outputs": [],
"source": [
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"outputs": []
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "nGnpSYAAh2bQ"
},
"source": [
"### Step-2: Set OpenAI environment variables\n",
"\n",
"You can find this env variable on your [OpenAI dashboard](https://platform.openai.com/account/api-keys)."
],
"metadata": {
"id": "nGnpSYAAh2bQ"
}
]
},
{
"cell_type": "code",
"source": [
"import os\n",
"from embedchain import App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
],
"execution_count": null,
"metadata": {
"id": "0fBdQ9GAiRvK"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"import os\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
]
},
{
"cell_type": "markdown",
"source": [
"### Step-3: Define your Vector Database config"
],
"metadata": {
"id": "Ns6RhPfbiitr"
}
},
"source": [
"### Step-3: Define your Vector Database config"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "S9CkxVjriotB"
},
"outputs": [],
"source": [
"config = \"\"\"\n",
"vectordb:\n",
@@ -95,64 +95,64 @@
"# Write the multi-line string to a YAML file\n",
"with open('chromadb.yaml', 'w') as file:\n",
" file.write(config)"
],
"metadata": {
"id": "S9CkxVjriotB"
},
"execution_count": null,
"outputs": []
]
},
{
"cell_type": "markdown",
"source": [
"### Step-4 Create embedchain app based on the config"
],
"metadata": {
"id": "PGt6uPLIi1CS"
}
},
"source": [
"### Step-4 Create embedchain app based on the config"
]
},
{
"cell_type": "code",
"source": [
"app = App.from_config(yaml_path=\"chromadb.yaml\")"
],
"execution_count": null,
"metadata": {
"id": "Amzxk3m-i3tD"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"app = App.from_config(yaml_path=\"chromadb.yaml\")"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-5: Add data sources to your app"
],
"metadata": {
"id": "XNXv4yZwi7ef"
}
},
"source": [
"### Step-5: Add data sources to your app"
]
},
{
"cell_type": "code",
"source": [
"app.add(\"https://www.forbes.com/profile/elon-musk\")"
],
"execution_count": null,
"metadata": {
"id": "Sn_0rx9QjIY9"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"app.add(\"https://www.forbes.com/profile/elon-musk\")"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-6: All set. Now start asking questions related to your data"
],
"metadata": {
"id": "_7W6fDeAjMAP"
}
},
"source": [
"### Step-6: All set. Now start asking questions related to your data"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cvIK7dWRjN_f"
},
"outputs": [],
"source": [
"while(True):\n",
" question = input(\"Enter question: \")\n",
@@ -160,12 +160,21 @@
" break\n",
" answer = app.query(question)\n",
" print(answer)"
],
"metadata": {
"id": "cvIK7dWRjN_f"
},
"execution_count": null,
"outputs": []
]
}
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "Python 3",
"name": "python3"
},
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 0
}
+10 -1
View File
@@ -33,6 +33,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -69,7 +78,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\"\n",
"os.environ[\"COHERE_API_KEY\"] = \"xxx\""
+92 -83
View File
@@ -1,95 +1,95 @@
{
"nbformat": 4,
"nbformat_minor": 0,
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"name": "python3",
"display_name": "Python 3"
},
"language_info": {
"name": "python"
}
},
"cells": [
{
"cell_type": "markdown",
"source": [
"## Cookbook for using ElasticSearchDB with Embedchain"
],
"metadata": {
"id": "b02n_zJ_hl3d"
}
},
"source": [
"## Cookbook for using ElasticSearchDB with Embedchain"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-1: Install embedchain package"
],
"metadata": {
"id": "gyJ6ui2vhtMY"
}
},
"source": [
"### Step-1: Install embedchain package"
]
},
{
"cell_type": "code",
"source": [
"!pip install embedchain"
],
"execution_count": null,
"metadata": {
"id": "-NbXjAdlh0vJ"
},
"outputs": [],
"source": [
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"outputs": []
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "nGnpSYAAh2bQ"
},
"source": [
"### Step-2: Set OpenAI environment variables and install the dependencies.\n",
"\n",
"You can find this env variable on your [OpenAI dashboard](https://platform.openai.com/account/api-keys). Now lets install the dependencies needed for Elasticsearch."
],
"metadata": {
"id": "nGnpSYAAh2bQ"
}
]
},
{
"cell_type": "code",
"source": [
"!pip install --upgrade 'embedchain[elasticsearch]'"
],
"execution_count": null,
"metadata": {
"id": "-MUFRfxV7Jk7"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"!pip install --upgrade 'embedchain[elasticsearch]'"
]
},
{
"cell_type": "code",
"source": [
"import os\n",
"from embedchain import App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
],
"execution_count": null,
"metadata": {
"id": "0fBdQ9GAiRvK"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"import os\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
]
},
{
"cell_type": "markdown",
"source": [
"### Step-3: Define your Vector Database config"
],
"metadata": {
"id": "Ns6RhPfbiitr"
}
},
"source": [
"### Step-3: Define your Vector Database config"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "S9CkxVjriotB"
},
"outputs": [],
"source": [
"config = \"\"\"\n",
"vectordb:\n",
@@ -104,64 +104,64 @@
"# Write the multi-line string to a YAML file\n",
"with open('elasticsearch.yaml', 'w') as file:\n",
" file.write(config)"
],
"metadata": {
"id": "S9CkxVjriotB"
},
"execution_count": null,
"outputs": []
]
},
{
"cell_type": "markdown",
"source": [
"### Step-4 Create embedchain app based on the config"
],
"metadata": {
"id": "PGt6uPLIi1CS"
}
},
"source": [
"### Step-4 Create embedchain app based on the config"
]
},
{
"cell_type": "code",
"source": [
"app = App.from_config(yaml_path=\"elasticsearch.yaml\")"
],
"execution_count": null,
"metadata": {
"id": "Amzxk3m-i3tD"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"app = App.from_config(yaml_path=\"elasticsearch.yaml\")"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-5: Add data sources to your app"
],
"metadata": {
"id": "XNXv4yZwi7ef"
}
},
"source": [
"### Step-5: Add data sources to your app"
]
},
{
"cell_type": "code",
"source": [
"app.add(\"https://www.forbes.com/profile/elon-musk\")"
],
"execution_count": null,
"metadata": {
"id": "Sn_0rx9QjIY9"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"app.add(\"https://www.forbes.com/profile/elon-musk\")"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-6: All set. Now start asking questions related to your data"
],
"metadata": {
"id": "_7W6fDeAjMAP"
}
},
"source": [
"### Step-6: All set. Now start asking questions related to your data"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cvIK7dWRjN_f"
},
"outputs": [],
"source": [
"while(True):\n",
" question = input(\"Enter question: \")\n",
@@ -169,12 +169,21 @@
" break\n",
" answer = app.query(question)\n",
" print(answer)"
],
"metadata": {
"id": "cvIK7dWRjN_f"
},
"execution_count": null,
"outputs": []
]
}
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "Python 3",
"name": "python3"
},
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 0
}
+1 -1
View File
@@ -33,7 +33,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"from embedchain.config import AppConfig\n",
"\n",
"\n",
+1 -1
View File
@@ -7,7 +7,7 @@
"metadata": {},
"outputs": [],
"source": [
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"embedchain_docs_bot = App()"
]
+10 -1
View File
@@ -33,6 +33,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -67,7 +76,7 @@
},
"outputs": [],
"source": [
"from embedchain import App"
"from embedchain import Pipeline as App"
]
},
{
+10 -1
View File
@@ -34,6 +34,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -84,7 +93,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"HUGGINGFACE_ACCESS_TOKEN\"] = \"hf_xxx\""
]
+10 -1
View File
@@ -34,6 +34,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -54,7 +63,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\"\n",
"os.environ[\"JINACHAT_API_KEY\"] = \"xxx\""
+10 -1
View File
@@ -33,6 +33,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -64,7 +73,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\"\n",
"os.environ[\"REPLICATE_API_TOKEN\"] = \"xxx\""
+10 -1
View File
@@ -34,6 +34,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -54,7 +63,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
]
+92 -83
View File
@@ -1,95 +1,95 @@
{
"nbformat": 4,
"nbformat_minor": 0,
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"name": "python3",
"display_name": "Python 3"
},
"language_info": {
"name": "python"
}
},
"cells": [
{
"cell_type": "markdown",
"source": [
"## Cookbook for using OpenSearchDB with Embedchain"
],
"metadata": {
"id": "b02n_zJ_hl3d"
}
},
"source": [
"## Cookbook for using OpenSearchDB with Embedchain"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-1: Install embedchain package"
],
"metadata": {
"id": "gyJ6ui2vhtMY"
}
},
"source": [
"### Step-1: Install embedchain package"
]
},
{
"cell_type": "code",
"source": [
"!pip install embedchain"
],
"execution_count": null,
"metadata": {
"id": "-NbXjAdlh0vJ"
},
"outputs": [],
"source": [
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"outputs": []
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "nGnpSYAAh2bQ"
},
"source": [
"### Step-2: Set OpenAI environment variables and install the dependencies.\n",
"\n",
"You can find this env variable on your [OpenAI dashboard](https://platform.openai.com/account/api-keys). Now lets install the dependencies needed for Opensearch."
],
"metadata": {
"id": "nGnpSYAAh2bQ"
}
]
},
{
"cell_type": "code",
"source": [
"!pip install --upgrade 'embedchain[opensearch]'"
],
"execution_count": null,
"metadata": {
"id": "-MUFRfxV7Jk7"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"!pip install --upgrade 'embedchain[opensearch]'"
]
},
{
"cell_type": "code",
"source": [
"import os\n",
"from embedchain import App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
],
"execution_count": null,
"metadata": {
"id": "0fBdQ9GAiRvK"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"import os\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
]
},
{
"cell_type": "markdown",
"source": [
"### Step-3: Define your Vector Database config"
],
"metadata": {
"id": "Ns6RhPfbiitr"
}
},
"source": [
"### Step-3: Define your Vector Database config"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "S9CkxVjriotB"
},
"outputs": [],
"source": [
"config = \"\"\"\n",
"vectordb:\n",
@@ -108,64 +108,64 @@
"# Write the multi-line string to a YAML file\n",
"with open('opensearch.yaml', 'w') as file:\n",
" file.write(config)"
],
"metadata": {
"id": "S9CkxVjriotB"
},
"execution_count": null,
"outputs": []
]
},
{
"cell_type": "markdown",
"source": [
"### Step-4 Create embedchain app based on the config"
],
"metadata": {
"id": "PGt6uPLIi1CS"
}
},
"source": [
"### Step-4 Create embedchain app based on the config"
]
},
{
"cell_type": "code",
"source": [
"app = App.from_config(yaml_path=\"opensearch.yaml\")"
],
"execution_count": null,
"metadata": {
"id": "Amzxk3m-i3tD"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"app = App.from_config(yaml_path=\"opensearch.yaml\")"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-5: Add data sources to your app"
],
"metadata": {
"id": "XNXv4yZwi7ef"
}
},
"source": [
"### Step-5: Add data sources to your app"
]
},
{
"cell_type": "code",
"source": [
"app.add(\"https://www.forbes.com/profile/elon-musk\")"
],
"execution_count": null,
"metadata": {
"id": "Sn_0rx9QjIY9"
},
"execution_count": null,
"outputs": []
"outputs": [],
"source": [
"app.add(\"https://www.forbes.com/profile/elon-musk\")"
]
},
{
"cell_type": "markdown",
"source": [
"### Step-6: All set. Now start asking questions related to your data"
],
"metadata": {
"id": "_7W6fDeAjMAP"
}
},
"source": [
"### Step-6: All set. Now start asking questions related to your data"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cvIK7dWRjN_f"
},
"outputs": [],
"source": [
"while(True):\n",
" question = input(\"Enter question: \")\n",
@@ -173,12 +173,21 @@
" break\n",
" answer = app.query(question)\n",
" print(answer)"
],
"metadata": {
"id": "cvIK7dWRjN_f"
},
"execution_count": null,
"outputs": []
]
}
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "Python 3",
"name": "python3"
},
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 0
}
+10 -1
View File
@@ -29,6 +29,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -60,7 +69,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\"\n",
"os.environ[\"PINECONE_API_KEY\"] = \"xxx\"\n",
+10 -1
View File
@@ -33,6 +33,15 @@
"!pip install embedchain"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install embedchain[dataloaders]"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -64,7 +73,7 @@
"outputs": [],
"source": [
"import os\n",
"from embedchain import App\n",
"from embedchain import Pipeline as App\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-xxx\""
]
Generated
+31 -87
View File
@@ -2538,86 +2538,28 @@ files = [
]
[[package]]
name = "jq"
version = "1.6.0"
description = "jq is a lightweight and flexible JSON processor."
optional = true
python-versions = ">=3.5"
name = "jsonpatch"
version = "1.33"
description = "Apply JSON-Patches (RFC 6902)"
optional = false
python-versions = ">=2.7, !=3.0.*, !=3.1.*, !=3.2.*, !=3.3.*, !=3.4.*, !=3.5.*, !=3.6.*"
files = [
{file = "jq-1.6.0-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:5773851cfb9ec6525f362f5bf7f18adab5c1fd1f0161c3599264cd0118c799da"},
{file = "jq-1.6.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:a758df4eae767a21ebd8466dfd0066d99c9741d9f7fd4a7e1d5b5227e1924af7"},
{file = "jq-1.6.0-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:15cf9dd3e7fb40d029f12f60cf418374c0b830a6ea6267dd285b48809069d6af"},
{file = "jq-1.6.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:c7e768cf5c25d703d944ef81c787d745da0eb266a97768f3003f91c4c828118d"},
{file = "jq-1.6.0-cp310-cp310-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:85a697b3cdc65e787f90faa1237caa44c117b6b2853f21263c3f0b16661b192c"},
{file = "jq-1.6.0-cp310-cp310-musllinux_1_1_aarch64.whl", hash = "sha256:944e081c328501ddc0a22a8f08196df72afe7910ca11e1a1f21244410dbdd3b3"},
{file = "jq-1.6.0-cp310-cp310-musllinux_1_1_i686.whl", hash = "sha256:09262d0e0cafb03acc968622e6450bb08abfb14c793bab47afd2732b47c655fd"},
{file = "jq-1.6.0-cp310-cp310-musllinux_1_1_x86_64.whl", hash = "sha256:611f460f616f957d57e0da52ac6e1e6294b073c72a89651da5546a31347817bd"},
{file = "jq-1.6.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:aba35b5cc07cd75202148e55f47ede3f4d0819b51c80f6d0c82a2ca47db07189"},
{file = "jq-1.6.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:ef5ddb76b03610df19a53583348aed3604f21d0ba6b583ee8d079e8df026cd47"},
{file = "jq-1.6.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:872f322ff7bfd7daff41b7e8248d414a88722df0e82d1027f3b091a438543e63"},
{file = "jq-1.6.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ca7a2982ff26f4620ac03099542a0230dabd8787af3f03ac93660598e26acbf0"},
{file = "jq-1.6.0-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:316affc6debf15eb05b7fd8e84ebf8993042b10b840e8d2a504659fb3ba07992"},
{file = "jq-1.6.0-cp311-cp311-musllinux_1_1_aarch64.whl", hash = "sha256:9bc42ade4de77fe4370c0e8e105ef10ad1821ef74d61dcc70982178b9ecfdc72"},
{file = "jq-1.6.0-cp311-cp311-musllinux_1_1_i686.whl", hash = "sha256:02da59230912b886ed45489f3693ce75877f3e99c9e490c0a2dbcf0db397e0df"},
{file = "jq-1.6.0-cp311-cp311-musllinux_1_1_x86_64.whl", hash = "sha256:7ea39f89aa469eb12145ddd686248916cd6d186647aa40b319af8444b1f45a2d"},
{file = "jq-1.6.0-cp312-cp312-macosx_10_9_x86_64.whl", hash = "sha256:6e9016f5ba064fabc527adb609ebae1f27cac20c8e0da990abae1cfb12eca706"},
{file = "jq-1.6.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:022be104a548f7fbddf103ce749937956df9d37a4f2f1650396dacad73bce7ee"},
{file = "jq-1.6.0-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:1d5a7f31f779e1aa3d165eaec237d74c7f5728227e81023a576c939ba3da34f8"},
{file = "jq-1.6.0-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:5f1533a2a15c42be3368878b4031b12f30441246878e0b5f6bedfdd7828cdb1f"},
{file = "jq-1.6.0-cp312-cp312-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:8aa67a304e58aa85c550ec011a68754ae49abe227b37d63a351feef4eea4c7a7"},
{file = "jq-1.6.0-cp312-cp312-musllinux_1_1_aarch64.whl", hash = "sha256:0893d1590cfa6facaf787cc6c28ac51e47d0d06a303613f84d4943ac0ca98e32"},
{file = "jq-1.6.0-cp312-cp312-musllinux_1_1_i686.whl", hash = "sha256:63db80b4803905a4f4f6c87a17aa1816c530f6262bc795773ebe60f8ab259092"},
{file = "jq-1.6.0-cp312-cp312-musllinux_1_1_x86_64.whl", hash = "sha256:e2c1f429e644cb962e846a6157b5352c3c556fbd0b22bba9fc2fea0710333369"},
{file = "jq-1.6.0-cp36-cp36m-macosx_10_9_x86_64.whl", hash = "sha256:bcf574f28809ec63b8df6456fdd4a981751b7466851e80621993b4e9d3e3c8ee"},
{file = "jq-1.6.0-cp36-cp36m-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:49dbe0f003b411ca52b5d0afaf09cad8e430a1011181c86f2ef720a0956f31c1"},
{file = "jq-1.6.0-cp36-cp36m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9f5a9c4185269a5faf395aa7ca086c7b02c9c8b448d542be3b899041d06e0970"},
{file = "jq-1.6.0-cp36-cp36m-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:8265f3badcd125f234e55dfc02a078c5decdc6faafcd453fde04d4c0d2699886"},
{file = "jq-1.6.0-cp36-cp36m-musllinux_1_1_aarch64.whl", hash = "sha256:c6c39b53d000d2f7f9f6338061942b83c9034d04f3bc99acae0867d23c9e7127"},
{file = "jq-1.6.0-cp36-cp36m-musllinux_1_1_i686.whl", hash = "sha256:9897931ea7b9a46f8165ee69737ece4a2e6dbc8e10ececb81f459d51d71401df"},
{file = "jq-1.6.0-cp36-cp36m-musllinux_1_1_x86_64.whl", hash = "sha256:6312237159e88e92775ea497e0c739590528062d4074544aacf12a08d252f966"},
{file = "jq-1.6.0-cp37-cp37m-macosx_10_9_x86_64.whl", hash = "sha256:aa786a60bdd1a3571f092a4021dd9abf6c46798530fa99f19ecf4f0fceaa7eaf"},
{file = "jq-1.6.0-cp37-cp37m-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:22495573d8221320d3433e1aeded40132bd8e1726845629558bd73aaa66eef7b"},
{file = "jq-1.6.0-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:711eabc5d33ef3ec581e0744d9cff52f43896d84847a2692c287a0140a29c915"},
{file = "jq-1.6.0-cp37-cp37m-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:57e75c1563d083b0424690b3c3ef2bb519e670770931fe633101ede16615d6ee"},
{file = "jq-1.6.0-cp37-cp37m-musllinux_1_1_aarch64.whl", hash = "sha256:c795f175b1a13bd716a0c180d062cc8e305271f47bbdb9eb0f0f62f7e4f5def4"},
{file = "jq-1.6.0-cp37-cp37m-musllinux_1_1_i686.whl", hash = "sha256:227b178b22a7f91ae88525810441791b1ca1fc71c86f03190911793be15cec3d"},
{file = "jq-1.6.0-cp37-cp37m-musllinux_1_1_x86_64.whl", hash = "sha256:780eb6383fbae12afa819ef676fc93e1548ae4b076c004a393af26a04b460742"},
{file = "jq-1.6.0-cp38-cp38-macosx_10_9_x86_64.whl", hash = "sha256:08ded6467f4ef89fec35b2bf310f210f8cd13fbd9d80e521500889edf8d22441"},
{file = "jq-1.6.0-cp38-cp38-macosx_11_0_arm64.whl", hash = "sha256:49e44ed677713f4115bd5bf2dbae23baa4cd503be350e12a1c1f506b0687848f"},
{file = "jq-1.6.0-cp38-cp38-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:984f33862af285ad3e41e23179ac4795f1701822473e1a26bf87ff023e5a89ea"},
{file = "jq-1.6.0-cp38-cp38-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f42264fafc6166efb5611b5d4cb01058887d050a6c19334f6a3f8a13bb369df5"},
{file = "jq-1.6.0-cp38-cp38-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:a67154f150aaf76cc1294032ed588436eb002097dd4fd1e283824bf753a05080"},
{file = "jq-1.6.0-cp38-cp38-musllinux_1_1_aarch64.whl", hash = "sha256:1b3b95d5fd20e51f18a42647fdb52e5d8aaf150b7a666dd659cf282a2221ee3f"},
{file = "jq-1.6.0-cp38-cp38-musllinux_1_1_i686.whl", hash = "sha256:3a8d98f72111043e75610cad7fa9ec5aec0b1ee2f7332dc7fd0f6603ea8144f8"},
{file = "jq-1.6.0-cp38-cp38-musllinux_1_1_x86_64.whl", hash = "sha256:487483f10ae8f70e6acf7723f31b329736de4b421ce56b2f43b46d5cbd7337b0"},
{file = "jq-1.6.0-cp39-cp39-macosx_10_9_x86_64.whl", hash = "sha256:18a700f55b7ef83a1382edf0a48cb176b22bacd155e097375ef2345ff8621d97"},
{file = "jq-1.6.0-cp39-cp39-macosx_11_0_arm64.whl", hash = "sha256:68aec8534ac3c4705e524b4ef54f66b8bdc867df9e0af2c3895e82c6774b5374"},
{file = "jq-1.6.0-cp39-cp39-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:b7a164748dbd03bb06d23bab7ead7ba7e5c4fcfebea7b082bdcd21d14136931e"},
{file = "jq-1.6.0-cp39-cp39-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:aa22d24740276a8ce82411e4960ed2b5fab476230f913f9d9cf726f766a22208"},
{file = "jq-1.6.0-cp39-cp39-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:4c1a6fae1b74b3e0478e281eb6addedad7b32421221ac685e21c1d49af5e997f"},
{file = "jq-1.6.0-cp39-cp39-musllinux_1_1_aarch64.whl", hash = "sha256:ce628546c22792b8870b9815086f65873ebb78d7bf617b5a16dd839adba36538"},
{file = "jq-1.6.0-cp39-cp39-musllinux_1_1_i686.whl", hash = "sha256:7bb685f337cf5d4f4fe210c46220e31a7baec02a0ca0df3ace3dd4780328fc30"},
{file = "jq-1.6.0-cp39-cp39-musllinux_1_1_x86_64.whl", hash = "sha256:bdbbc509a35ee6082d79c1f25eb97c08f1c59043d21e0772cd24baa909505899"},
{file = "jq-1.6.0-pp310-pypy310_pp73-macosx_10_9_x86_64.whl", hash = "sha256:1b332dfdf0d81fb7faf3d12aabf997565d7544bec9812e0ac5ee55e60ef4df8c"},
{file = "jq-1.6.0-pp310-pypy310_pp73-macosx_11_0_arm64.whl", hash = "sha256:3a4f6ef8c0bd19beae56074c50026665d66345d1908f050e5c442ceac2efe398"},
{file = "jq-1.6.0-pp310-pypy310_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:5184c2fcca40f8f2ab1b14662721accf68b4b5e772e2f5336fec24aa58fe235a"},
{file = "jq-1.6.0-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:689429fe1e07a2d6041daba2c21ced3a24895b2745326deb0c90ccab9386e116"},
{file = "jq-1.6.0-pp310-pypy310_pp73-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:8405d1c996c83711570f16aac32e3bf2c116d6fa4254a820276b87aed544d7e8"},
{file = "jq-1.6.0-pp37-pypy37_pp73-macosx_10_9_x86_64.whl", hash = "sha256:138d56c7efc8bb162c1cfc3806bd6b4d779115943af36c9e3b8ca644dde856c2"},
{file = "jq-1.6.0-pp37-pypy37_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:fd28f8395687e45bba56dc771284ebb6492b02037f74f450176c102f3f4e86a3"},
{file = "jq-1.6.0-pp37-pypy37_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:b2c783288bf10e67aad321b58735e663f4975d7ddfbfb0a5bca8428eee283bde"},
{file = "jq-1.6.0-pp37-pypy37_pp73-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:206391ac5b2eb556720b94f0f131558cbf8d82d8cc7e0404e733eeef48bcd823"},
{file = "jq-1.6.0-pp38-pypy38_pp73-macosx_10_9_x86_64.whl", hash = "sha256:35090fea1283402abc3a13b43261468162199d8b5dcdaba2d1029e557ed23070"},
{file = "jq-1.6.0-pp38-pypy38_pp73-macosx_11_0_arm64.whl", hash = "sha256:201c6384603aec87a744ad7b393cc4f1c58ece23d6e0a6c216a47bfcc405d231"},
{file = "jq-1.6.0-pp38-pypy38_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a3d8b075351c29653f29a1fec5d31bc88aa198a0843c0a9550b9be74d8fab33b"},
{file = "jq-1.6.0-pp38-pypy38_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:132e41f6e988c42b91c04b1b60dd8fa185a5c0681de5438ea1e6c64f5329768c"},
{file = "jq-1.6.0-pp38-pypy38_pp73-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:e1cb4751808b1d0dbddd37319e0c574fb0c3a29910d52ba35890b1343a1f1e59"},
{file = "jq-1.6.0-pp39-pypy39_pp73-macosx_10_9_x86_64.whl", hash = "sha256:bd158911ed5f5c644f557ad94d6424c411560632a885eae47d105f290f0109cb"},
{file = "jq-1.6.0-pp39-pypy39_pp73-macosx_11_0_arm64.whl", hash = "sha256:64bc09ae6a9d9b82b78e15d142f90b816228bd3ee48833ddca3ff8c08e163fa7"},
{file = "jq-1.6.0-pp39-pypy39_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:f4eed167322662f4b7e65235723c54aa6879f6175b6f9b68bc24887549637ffb"},
{file = "jq-1.6.0-pp39-pypy39_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:64bb4b305e2fabe5b5161b599bf934aceb0e0e7d3dd8f79246737ea91a2bc9ae"},
{file = "jq-1.6.0-pp39-pypy39_pp73-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:165bfbe29bf73878d073edf75f384b7da8a9657ba0ab9fb1e5fe6be65ab7debb"},
{file = "jq-1.6.0.tar.gz", hash = "sha256:c7711f0c913a826a00990736efa6ffc285f8ef433414516bb14b7df971d6c1ea"},
{file = "jsonpatch-1.33-py2.py3-none-any.whl", hash = "sha256:0ae28c0cd062bbd8b8ecc26d7d164fbbea9652a1a3693f3b956c1eae5145dade"},
{file = "jsonpatch-1.33.tar.gz", hash = "sha256:9fcd4009c41e6d12348b4a0ff2563ba56a2923a7dfee731d004e212e1ee5030c"},
]
[package.dependencies]
jsonpointer = ">=1.9"
[[package]]
name = "jsonpointer"
version = "2.4"
description = "Identify specific nodes in a JSON document (RFC 6901)"
optional = false
python-versions = ">=2.7, !=3.0.*, !=3.1.*, !=3.2.*, !=3.3.*, !=3.4.*, !=3.5.*, !=3.6.*"
files = [
{file = "jsonpointer-2.4-py2.py3-none-any.whl", hash = "sha256:15d51bba20eea3165644553647711d150376234112651b4f1811022aecad7d7a"},
{file = "jsonpointer-2.4.tar.gz", hash = "sha256:585cee82b70211fa9e6043b7bb89db6e1aa49524340dde8ad6b63206ea689d88"},
]
[[package]]
@@ -2735,20 +2677,22 @@ files = [
[[package]]
name = "langchain"
version = "0.0.279"
version = "0.0.303"
description = "Building applications with LLMs through composability"
optional = false
python-versions = ">=3.8.1,<4.0"
files = [
{file = "langchain-0.0.279-py3-none-any.whl", hash = "sha256:d91e65b3a210e9b52e42ed67c65d76caeaa3195ff097217f5ab2ca477bd3ae27"},
{file = "langchain-0.0.279.tar.gz", hash = "sha256:344e33d73c76ae35cdd8c3ba16c965bac29e0e889c93064718c5982002cbe30c"},
{file = "langchain-0.0.303-py3-none-any.whl", hash = "sha256:1745961f66b60bc3b513820a34c560dd37c4ba4b7499ba82545dc4816d0133bd"},
{file = "langchain-0.0.303.tar.gz", hash = "sha256:84d2727eb8b3b27a9d0aa0da9f05408c2564a4a923c7d5b154a16e488430e725"},
]
[package.dependencies]
aiohttp = ">=3.8.3,<4.0.0"
anyio = "<4.0"
async-timeout = {version = ">=4.0.0,<5.0.0", markers = "python_version < \"3.11\""}
dataclasses-json = ">=0.5.7,<0.6.0"
langsmith = ">=0.0.21,<0.1.0"
dataclasses-json = ">=0.5.7,<0.7"
jsonpatch = ">=1.33,<2.0"
langsmith = ">=0.0.38,<0.1.0"
numexpr = ">=2.8.4,<3.0.0"
numpy = ">=1,<2"
pydantic = ">=1,<3"
@@ -2764,7 +2708,7 @@ clarifai = ["clarifai (>=9.1.0)"]
cohere = ["cohere (>=4,<5)"]
docarray = ["docarray[hnswlib] (>=0.32.0,<0.33.0)"]
embeddings = ["sentence-transformers (>=2,<3)"]
extended-testing = ["amazon-textract-caller (<2)", "assemblyai (>=0.17.0,<0.18.0)", "atlassian-python-api (>=3.36.0,<4.0.0)", "beautifulsoup4 (>=4,<5)", "bibtexparser (>=1.4.0,<2.0.0)", "cassio (>=0.0.7,<0.0.8)", "chardet (>=5.1.0,<6.0.0)", "esprima (>=4.0.1,<5.0.0)", "faiss-cpu (>=1,<2)", "feedparser (>=6.0.10,<7.0.0)", "geopandas (>=0.13.1,<0.14.0)", "gitpython (>=3.1.32,<4.0.0)", "gql (>=3.4.1,<4.0.0)", "html2text (>=2020.1.16,<2021.0.0)", "jinja2 (>=3,<4)", "jq (>=1.4.1,<2.0.0)", "lxml (>=4.9.2,<5.0.0)", "markdownify (>=0.11.6,<0.12.0)", "mwparserfromhell (>=0.6.4,<0.7.0)", "mwxml (>=0.3.3,<0.4.0)", "newspaper3k (>=0.2.8,<0.3.0)", "openai (>=0,<1)", "openapi-schema-pydantic (>=1.2,<2.0)", "pandas (>=2.0.1,<3.0.0)", "pdfminer-six (>=20221105,<20221106)", "pgvector (>=0.1.6,<0.2.0)", "psychicapi (>=0.8.0,<0.9.0)", "py-trello (>=0.19.0,<0.20.0)", "pymupdf (>=1.22.3,<2.0.0)", "pypdf (>=3.4.0,<4.0.0)", "pypdfium2 (>=4.10.0,<5.0.0)", "pyspark (>=3.4.0,<4.0.0)", "rank-bm25 (>=0.2.2,<0.3.0)", "rapidfuzz (>=3.1.1,<4.0.0)", "requests-toolbelt (>=1.0.0,<2.0.0)", "scikit-learn (>=1.2.2,<2.0.0)", "sqlite-vss (>=0.1.2,<0.2.0)", "streamlit (>=1.18.0,<2.0.0)", "sympy (>=1.12,<2.0)", "telethon (>=1.28.5,<2.0.0)", "tqdm (>=4.48.0)", "xata (>=1.0.0a7,<2.0.0)", "xmltodict (>=0.13.0,<0.14.0)"]
extended-testing = ["amazon-textract-caller (<2)", "assemblyai (>=0.17.0,<0.18.0)", "atlassian-python-api (>=3.36.0,<4.0.0)", "beautifulsoup4 (>=4,<5)", "bibtexparser (>=1.4.0,<2.0.0)", "cassio (>=0.1.0,<0.2.0)", "chardet (>=5.1.0,<6.0.0)", "dashvector (>=1.0.1,<2.0.0)", "esprima (>=4.0.1,<5.0.0)", "faiss-cpu (>=1,<2)", "feedparser (>=6.0.10,<7.0.0)", "geopandas (>=0.13.1,<0.14.0)", "gitpython (>=3.1.32,<4.0.0)", "gql (>=3.4.1,<4.0.0)", "html2text (>=2020.1.16,<2021.0.0)", "jinja2 (>=3,<4)", "jq (>=1.4.1,<2.0.0)", "lxml (>=4.9.2,<5.0.0)", "markdownify (>=0.11.6,<0.12.0)", "mwparserfromhell (>=0.6.4,<0.7.0)", "mwxml (>=0.3.3,<0.4.0)", "newspaper3k (>=0.2.8,<0.3.0)", "openai (>=0,<1)", "openapi-schema-pydantic (>=1.2,<2.0)", "pandas (>=2.0.1,<3.0.0)", "pdfminer-six (>=20221105,<20221106)", "pgvector (>=0.1.6,<0.2.0)", "psychicapi (>=0.8.0,<0.9.0)", "py-trello (>=0.19.0,<0.20.0)", "pymupdf (>=1.22.3,<2.0.0)", "pypdf (>=3.4.0,<4.0.0)", "pypdfium2 (>=4.10.0,<5.0.0)", "pyspark (>=3.4.0,<4.0.0)", "rank-bm25 (>=0.2.2,<0.3.0)", "rapidfuzz (>=3.1.1,<4.0.0)", "requests-toolbelt (>=1.0.0,<2.0.0)", "scikit-learn (>=1.2.2,<2.0.0)", "sqlite-vss (>=0.1.2,<0.2.0)", "streamlit (>=1.18.0,<2.0.0)", "sympy (>=1.12,<2.0)", "telethon (>=1.28.5,<2.0.0)", "timescale-vector (>=0.0.1,<0.0.2)", "tqdm (>=4.48.0)", "xata (>=1.0.0a7,<2.0.0)", "xmltodict (>=0.13.0,<0.14.0)"]
javascript = ["esprima (>=4.0.1,<5.0.0)"]
llms = ["clarifai (>=9.1.0)", "cohere (>=4,<5)", "huggingface_hub (>=0,<1)", "manifest-ml (>=0.0.1,<0.0.2)", "nlpcloud (>=1,<2)", "openai (>=0,<1)", "openlm (>=0.0.5,<0.0.6)", "torch (>=1,<3)", "transformers (>=4,<5)"]
openai = ["openai (>=0,<1)", "tiktoken (>=0.3.2,<0.4.0)"]
@@ -7174,7 +7118,7 @@ testing = ["big-O", "jaraco.functools", "jaraco.itertools", "more-itertools", "p
[extras]
cohere = ["cohere"]
community = ["llama-hub"]
dataloaders = ["beautifulsoup4", "docx2txt", "duckduckgo-search", "jq", "jq", "pypdf", "pytube", "sentence-transformers", "unstructured"]
dataloaders = ["beautifulsoup4", "docx2txt", "duckduckgo-search", "pypdf", "pytube", "sentence-transformers", "unstructured"]
discord = ["discord"]
elasticsearch = ["elasticsearch"]
gmail = ["llama-hub", "requests"]
@@ -7196,4 +7140,4 @@ whatsapp = ["flask", "twilio"]
[metadata]
lock-version = "2.0"
python-versions = ">=3.9,<3.13"
content-hash = "4021d63d76a9128c8d7342bc482dfe90fa9878fde1359c2f0730fa9188e503af"
content-hash = "2e5140f157ad4b3ad47f5dfcfe3ad98a2a40f68a5310cec2ca4245d2ce6af0fc"
+15 -16
View File
@@ -1,6 +1,6 @@
[tool.poetry]
name = "embedchain"
version = "0.0.82"
version = "0.0.88"
description = "Data platform for LLMs - Load, index, retrieve and sync any unstructured data"
authors = [
"Taranjeet Singh <taranjeet@embedchain.ai>",
@@ -90,16 +90,17 @@ color = true
[tool.poetry.dependencies]
python = ">=3.9,<3.13"
python-dotenv = "^1.0.0"
langchain = "^0.0.279"
langchain = "^0.0.303"
requests = "^2.31.0"
openai = ">=0.28.0"
tiktoken = { version="^0.4.0", optional=true }
chromadb ="^0.4.8"
youtube-transcript-api = { version="^0.6.1", optional=true }
beautifulsoup4 = { version="^4.12.2", optional=true }
pypdf = { version="^3.11.0", optional=true }
pytube = { version="^15.0.0", optional=true }
duckduckgo-search = { version="^3.8.5", optional=true }
chromadb = "^0.4.8"
posthog = "^3.0.2"
tiktoken = { version = "^0.4.0", optional = true }
youtube-transcript-api = { version = "^0.6.1", optional = true }
beautifulsoup4 = { version = "^4.12.2", optional = true }
pypdf = { version = "^3.11.0", optional = true }
pytube = { version = "^15.0.0", optional = true }
duckduckgo-search = { version = "^3.8.5", optional = true }
llama-hub = { version = "^0.0.29", optional = true }
sentence-transformers = { version = "^2.2.2", optional = true }
torch = { version = "2.0.0", optional = true }
@@ -113,12 +114,12 @@ twilio = { version = "^8.5.0", optional = true }
fastapi-poe = { version = "0.0.16", optional = true }
discord = { version = "^2.3.2", optional = true }
slack-sdk = { version = "3.21.3", optional = true }
cohere = { version = "^4.27", optional= true }
weaviate-client = { version = "^3.24.1", optional= true }
docx2txt = { version="^0.8", optional=true }
cohere = { version = "^4.27", optional = true }
weaviate-client = { version = "^3.24.1", optional = true }
docx2txt = { version = "^0.8", optional = true }
pinecone-client = { version = "^2.2.4", optional = true }
qdrant-client = { version = "1.6.3", optional = true }
unstructured = {extras = ["local-inference"], version = "^0.10.18", optional=true}
unstructured = {extras = ["local-inference"], version = "^0.10.18", optional = true}
pillow = { version = "10.0.1", optional = true }
torchvision = { version = ">=0.15.1, !=0.15.2", optional = true }
ftfy = { version = "6.1.1", optional = true }
@@ -127,7 +128,6 @@ huggingface_hub = { version = "^0.17.3", optional = true }
pymilvus = { version = "2.3.1", optional = true }
google-cloud-aiplatform = { version = "^1.26.1", optional = true }
replicate = { version = "^0.15.4", optional = true }
jq = { version=">=1.6.0", optional = true}
[tool.poetry.group.dev.dependencies]
black = "^23.3.0"
@@ -165,12 +165,10 @@ dataloaders=[
"beautifulsoup4",
"docx2txt",
"duckduckgo-search",
"jq",
"pypdf",
"pytube",
"sentence-transformers",
"unstructured",
"jq",
]
vertexai = ["google-cloud-aiplatform"]
llama2 = ["replicate"]
@@ -183,6 +181,7 @@ gmail = [
"google-auth-httplib2",
"google-api-core",
]
json = ["llama-hub"]
[tool.poetry.group.docs.dependencies]
+7
View File
@@ -14,3 +14,10 @@ def setup():
clean_db()
yield
clean_db()
@pytest.fixture(autouse=True)
def disable_telemetry():
os.environ["EC_TELEMETRY"] = "false"
yield
del os.environ["EC_TELEMETRY"]
+79 -20
View File
@@ -1,32 +1,91 @@
import hashlib
from unittest.mock import patch
from langchain.docstore.document import Document
from langchain.document_loaders.json_loader import \
JSONLoader as LangchainJSONLoader
import pytest
from llama_index.readers.schema.base import Document
from embedchain.loaders.json import JSONLoader
def test_load_data():
mock_document = [
Document(page_content="content1", metadata={"seq_num": 1}),
Document(page_content="content2", metadata={"seq_num": 2}),
def test_load_data(mocker):
content = "temp.json"
mock_document = {
"doc_id": hashlib.sha256((content + ", ".join(["content1", "content2"])).encode()).hexdigest(),
"data": [
{"content": "content1", "meta_data": {"url": content}},
{"content": "content2", "meta_data": {"url": content}},
],
}
mocker.patch("embedchain.loaders.json.JSONLoader.load_data", return_value=mock_document)
json_loader = JSONLoader()
result = json_loader.load_data(content)
assert "doc_id" in result
assert "data" in result
expected_data = [
{"content": "content1", "meta_data": {"url": content}},
{"content": "content2", "meta_data": {"url": content}},
]
with patch.object(LangchainJSONLoader, "load", return_value=mock_document):
content = "temp.json"
result = JSONLoader.load_data(content)
assert result["data"] == expected_data
assert "doc_id" in result
assert "data" in result
expected_doc_id = hashlib.sha256((content + ", ".join(["content1", "content2"])).encode()).hexdigest()
assert result["doc_id"] == expected_doc_id
expected_data = [
{"content": "content1", "meta_data": {"url": content, "row": 1}},
{"content": "content2", "meta_data": {"url": content, "row": 2}},
]
assert result["data"] == expected_data
def test_load_data_url(mocker):
content = "https://example.com/posts.json"
expected_doc_id = hashlib.sha256((content + ", ".join(["content1", "content2"])).encode()).hexdigest()
assert result["doc_id"] == expected_doc_id
mocker.patch("os.path.isfile", return_value=False) # Mocking os.path.isfile to simulate a URL case
mocker.patch(
"llama_hub.jsondata.base.JSONDataReader.load_data",
return_value=[Document(text="content1"), Document(text="content2")],
)
mock_response = mocker.Mock()
mock_response.status_code = 200
mock_response.json.return_value = {"document1": "content1", "document2": "content2"}
mocker.patch("requests.get", return_value=mock_response)
result = JSONLoader.load_data(content)
assert "doc_id" in result
assert "data" in result
expected_data = [
{"content": "content1", "meta_data": {"url": content}},
{"content": "content2", "meta_data": {"url": content}},
]
assert result["data"] == expected_data
expected_doc_id = hashlib.sha256((content + ", ".join(["content1", "content2"])).encode()).hexdigest()
assert result["doc_id"] == expected_doc_id
def test_load_data_invalid_content(mocker):
mocker.patch("os.path.isfile", return_value=False)
mocker.patch("requests.get")
content = "123"
with pytest.raises(ValueError, match="Invalid content to load json data from"):
JSONLoader.load_data(content)
def test_load_data_invalid_url(mocker):
mocker.patch("os.path.isfile", return_value=False)
mock_response = mocker.Mock()
mock_response.status_code = 404
mocker.patch("requests.get", return_value=mock_response)
content = "http://invalid-url.com/"
with pytest.raises(ValueError, match=f"Invalid content to load json data from: {content}"):
JSONLoader.load_data(content)
+63
View File
@@ -0,0 +1,63 @@
import logging
import os
from embedchain.telemetry.posthog import AnonymousTelemetry
class TestAnonymousTelemetry:
def test_init(self, mocker):
# Enable telemetry specifically for this test
os.environ["EC_TELEMETRY"] = "true"
mock_posthog = mocker.patch("embedchain.telemetry.posthog.Posthog")
telemetry = AnonymousTelemetry()
assert telemetry.project_api_key == "phc_PHQDA5KwztijnSojsxJ2c1DuJd52QCzJzT2xnSGvjN2"
assert telemetry.host == "https://app.posthog.com"
assert telemetry.enabled is True
assert telemetry.user_id
mock_posthog.assert_called_once_with(project_api_key=telemetry.project_api_key, host=telemetry.host)
def test_init_with_disabled_telemetry(self, mocker, monkeypatch):
mocker.patch("embedchain.telemetry.posthog.Posthog")
telemetry = AnonymousTelemetry()
assert telemetry.enabled is False
assert telemetry.posthog.disabled is True
def test_get_user_id(self, mocker, tmpdir):
mock_uuid = mocker.patch("embedchain.telemetry.posthog.uuid.uuid4")
mock_uuid.return_value = "unique_user_id"
config_file = tmpdir.join("config.json")
mocker.patch("embedchain.telemetry.posthog.CONFIG_FILE", str(config_file))
telemetry = AnonymousTelemetry()
user_id = telemetry.get_user_id()
assert user_id == "unique_user_id"
assert config_file.read() == '{"user_id": "unique_user_id"}'
def test_capture(self, mocker):
# Enable telemetry specifically for this test
os.environ["EC_TELEMETRY"] = "true"
mock_posthog = mocker.patch("embedchain.telemetry.posthog.Posthog")
telemetry = AnonymousTelemetry()
event_name = "test_event"
properties = {"key": "value"}
telemetry.capture(event_name, properties)
mock_posthog.assert_called_once_with(
project_api_key=telemetry.project_api_key,
host=telemetry.host,
)
mock_posthog.return_value.capture.assert_called_once_with(
telemetry.user_id,
event_name,
properties,
)
def test_capture_with_exception(self, mocker, caplog):
mock_posthog = mocker.patch("embedchain.telemetry.posthog.Posthog")
mock_posthog.return_value.capture.side_effect = Exception("Test Exception")
telemetry = AnonymousTelemetry()
event_name = "test_event"
properties = {"key": "value"}
with caplog.at_level(logging.ERROR):
telemetry.capture(event_name, properties)
assert "Failed to send telemetry event" in caplog.text
+1 -1
View File
@@ -76,7 +76,7 @@ class TestQdrantDB(unittest.TestCase):
qdrant_client_mock.return_value.upsert.assert_called_once_with(
collection_name="embedchain-store-1526",
points=Batch(
ids=["def", "ghi"],
ids=["abc", "def"],
payloads=[
{
"identifier": "123",