Compare commits
567 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| fb5a3bfd95 | |||
| 7441f1462d | |||
| c9240e7ca6 | |||
| 1e7618dfa4 | |||
| 4e5d34103f | |||
| da435bc025 | |||
| 2a43aa6902 | |||
| b620f8fae3 | |||
| 03f787d5cb | |||
| 19637804b3 | |||
| 80f145fceb | |||
| 34477d4936 | |||
| 4ec51f2dd6 | |||
| f842a92e25 | |||
| 83e8c97295 | |||
| 1a5d0d236a | |||
| ebbf90f4aa | |||
| 4f119692f1 | |||
| bbe56107fb | |||
| bd654e7aac | |||
| 33500a7ce2 | |||
| 4880557d51 | |||
| ea09b5f7f0 | |||
| 5258fd91ea | |||
| b305d674de | |||
| 7c24601d0f | |||
| 50c0285cb2 | |||
| 0a78198bb5 | |||
| edaeb78ccf | |||
| f80be2d2ea | |||
| 8700165b42 | |||
| 18fb92f1f8 | |||
| 14fc6bbadd | |||
| 5070a1d83e | |||
| 8a9088ea9d | |||
| 48b24f6f12 | |||
| f6ddd5ffc5 | |||
| b43a116b3c | |||
| 50512a5f03 | |||
| e3e107b31d | |||
| 21a04541ea | |||
| cdd5d8ac76 | |||
| 11094f504e | |||
| 5acaae5f56 | |||
| 4547d870af | |||
| dc0d8e0932 | |||
| c558eae9ce | |||
| abb9af66a6 | |||
| 4800e0344c | |||
| 439b425c61 | |||
| 2855f1635b | |||
| 08b67b4a78 | |||
| 1bddd46ed2 | |||
| 6ecdadfd97 | |||
| 4119040005 | |||
| 873eef6ef8 | |||
| 445fed4d3f | |||
| 52fd3e0dd4 | |||
| 8fd0e1f3b0 | |||
| 11fc4a8451 | |||
| e22293294e | |||
| 73e53aaff1 | |||
| 6fa946557f | |||
| fb0852f585 | |||
| 4070fc1bf0 | |||
| 00c1fa1ec7 | |||
| 04e77ef34e | |||
| 827d63d115 | |||
| e0d0f6e94c | |||
| fd07513004 | |||
| b0e436d9c4 | |||
| a4bfd9cfc6 | |||
| 8ca01918e5 | |||
| a5b2381458 | |||
| 26c771503b | |||
| 622ed4a7c9 | |||
| 940f0128d5 | |||
| 1354747ca8 | |||
| 9544c69c55 | |||
| 9ba445e623 | |||
| ebc5e25f98 | |||
| 78301ee63d | |||
| 797dea1dca | |||
| a0ff764f0a | |||
| a795798156 | |||
| 1a66f961f4 | |||
| 6fb2048af0 | |||
| ba9f186fc5 | |||
| 6c32d287b5 | |||
| 536f85b78a | |||
| f8619870ad | |||
| d00a2085d5 | |||
| 85ec61335a | |||
| 9b48a12c27 | |||
| c181ccbe42 | |||
| 8520033d44 | |||
| ebdce87fde | |||
| f2122ed696 | |||
| 3616eaadb4 | |||
| ef69c91b60 | |||
| 117824b32c | |||
| f77f5b996e | |||
| a4d32aec24 | |||
| 9111495fae | |||
| ee1e3f0957 | |||
| 4dc5c7348f | |||
| 4428768eaa | |||
| 11f4ce8fb6 | |||
| 6078738d34 | |||
| faacfeb891 | |||
| 8d7e8b6fb9 | |||
| 7e1d2ffdd7 | |||
| 91044ec591 | |||
| c77a75dfb5 | |||
| 6518c0c06b | |||
| 09cdaff9a2 | |||
| 56bf33ab7f | |||
| 752f638cfc | |||
| 92dd7edb57 | |||
| b4bb4cf053 | |||
| f0400e928a | |||
| aa5ad625af | |||
| f8f69eab03 | |||
| 2b2263acaa | |||
| 5e2e7fb639 | |||
| 6c12bc9044 | |||
| 9a11683003 | |||
| 38b4e06963 | |||
| 0766a44ccf | |||
| 036bf3a161 | |||
| 41bd258b93 | |||
| 38e212c721 | |||
| 2f285ea00a | |||
| d38120c839 | |||
| d94aee812b | |||
| 68d650ec40 | |||
| 769d926f5a | |||
| 9478bab04e | |||
| 7ad4af250f | |||
| 9fa368b114 | |||
| 4afef04f26 | |||
| 8fe2c3effc | |||
| fa78c972be | |||
| 0e66261644 | |||
| 819650a254 | |||
| 34c41c87dc | |||
| 2985b667b0 | |||
| 31bb0e7f0f | |||
| 8f28264aec | |||
| ec4fb11aa5 | |||
| b210723de1 | |||
| 433f99dd78 | |||
| e75c05112e | |||
| d2a5b50ff8 | |||
| 120690afd4 | |||
| 344dbeee42 | |||
| 3fe3b0320a | |||
| 75896b647f | |||
| 446d0975aa | |||
| b7d365119c | |||
| 2d9fbd4e49 | |||
| 22e14b5e65 | |||
| 1a654beea4 | |||
| f50f8a444a | |||
| 3cc3a0058d | |||
| ae473b5e3c | |||
| efb7e31565 | |||
| 069d265338 | |||
| 751a3a4bd1 | |||
| cb0499407e | |||
| 9afc6878c8 | |||
| 0b5b12575a | |||
| d79d30bf0c | |||
| 59600e2a5b | |||
| e572b5a3dc | |||
| 5b46daaee4 | |||
| 2784bae772 | |||
| 325e11f0de | |||
| 7444f59e3c | |||
| affe319460 | |||
| 862ff6cca6 | |||
| c020e65a50 | |||
| f582c1fe25 | |||
| 785929c502 | |||
| 68ec6615b1 | |||
| e2cca61cd3 | |||
| 69e83adae0 | |||
| 9e24aee40d | |||
| 3cff5e9898 | |||
| f3553040bc | |||
| 2b13984e11 | |||
| 0de9491c61 | |||
| c9df7a2020 | |||
| a7222e8c50 | |||
| 0373fa231c | |||
| 5f653e69ae | |||
| 2496ed133e | |||
| 62c0c52e31 | |||
| e36198dcc2 | |||
| 5fa6221f91 | |||
| f7696d1dc1 | |||
| 1878f8d4fc | |||
| 6c69ddef9b | |||
| 0c45020d81 | |||
| 4dfce44c1a | |||
| 1b661bb2fd | |||
| 73e726f6e3 | |||
| f58bbeffce | |||
| 99261e5fb5 | |||
| b4a59d1bd5 | |||
| 5c1f78879f | |||
| 94ba82f2a2 | |||
| b4ec14382b | |||
| 38ad57a22c | |||
| 60bbc180ba | |||
| a67d902b85 | |||
| ae2e9cb890 | |||
| 1976d38b25 | |||
| f5e3410e9a | |||
| 2f6ba642c7 | |||
| dd9b72dc62 | |||
| dd258c14b5 | |||
| 295cd3fac6 | |||
| c62663f2e4 | |||
| 367d6b70e2 | |||
| 27236bd1b2 | |||
| 6a82eb4287 | |||
| bd88fe3980 | |||
| 4f70fea6df | |||
| aee5bbb44b | |||
| a304ded500 | |||
| a54dde0509 | |||
| 52b4577d3b | |||
| e199f57279 | |||
| dec12b33a6 | |||
| 04daa1b206 | |||
| a7e1520d08 | |||
| 9e2b232c13 | |||
| 404e73af77 | |||
| a544b4d3ff | |||
| 6df63d9ca7 | |||
| 904baac153 | |||
| a926bcc640 | |||
| c0aafd38c9 | |||
| 19d80914df | |||
| d9d529987e | |||
| 12e6eaf802 | |||
| 7a026ea282 | |||
| 97dc90169b | |||
| 64c02f374a | |||
| 6c1ea7799e | |||
| 6be29f5bed | |||
| 68737da7a2 | |||
| da388b679f | |||
| 11f0d719f5 | |||
| e90673ae5b | |||
| f055028c6b | |||
| 106a338371 | |||
| 9fe80c5cca | |||
| 050706e95e | |||
| 6d2389de1c | |||
| dd97fad5a4 | |||
| 0f73ba9677 | |||
| 210fe9bb80 | |||
| ec8549d0e1 | |||
| b77d9d750f | |||
| a10823d309 | |||
| 3a09c2bd62 | |||
| a1394ce32e | |||
| 1020a4121f | |||
| 737837ae0b | |||
| 7ee2d0653b | |||
| 43926fb527 | |||
| 7bcc9e35dd | |||
| 6437661837 | |||
| b5f84f27ff | |||
| 48c38b5dc3 | |||
| b4f3bbbbc9 | |||
| 3cd50c4cd9 | |||
| cd2c40a9c4 | |||
| 33dcfe42b5 | |||
| db37b2ac15 | |||
| bee4e834b1 | |||
| 0272459435 | |||
| c0b5e93967 | |||
| 6983ebba49 | |||
| 9943d1e015 | |||
| b348251484 | |||
| e719b5bac3 | |||
| 54f43215cd | |||
| b246d9823e | |||
| 65c8dd445b | |||
| 0efbc80ac9 | |||
| 151746beec | |||
| c0ee680546 | |||
| 9303a1bf81 | |||
| d54cdc5b00 | |||
| b7a44ef472 | |||
| ae6f866901 | |||
| 7910cee259 | |||
| d66e647f99 | |||
| ff4a333be7 | |||
| 111749a95d | |||
| adde398b65 | |||
| d8897ce356 | |||
| 0ea8ab228c | |||
| d62a23edf6 | |||
| 51ebf3439b | |||
| 4a5ed1dd8d | |||
| e84b5034ea | |||
| a4831d6ed9 | |||
| 51b4966801 | |||
| 1d4e00ccef | |||
| c9fbc2e7d6 | |||
| fa34788df6 | |||
| 0f4f220119 | |||
| 512cfc9466 | |||
| 541b1cb7c7 | |||
| 36af1a7615 | |||
| b02e8feeda | |||
| 406c46e7f4 | |||
| e35eaf1bfc | |||
| 38426a7af1 | |||
| 141a23fb1e | |||
| bb28569abf | |||
| 1df46b2bb3 | |||
| 58f72e1ffe | |||
| 33409140b4 | |||
| f6b80e01a1 | |||
| 798d3fcc5a | |||
| 85f3ac428b | |||
| 9fcf2130b5 | |||
| 51df00729e | |||
| 023a61446f | |||
| e0b73e6a5a | |||
| 28460f725c | |||
| c93e49d2b8 | |||
| 07fb6bee54 | |||
| 3fa7db8420 | |||
| c14bd7b73b | |||
| 5201beaab0 | |||
| 122313d8a5 | |||
| 82fd595306 | |||
| 95c0d47236 | |||
| 919cc74e94 | |||
| d839991acb | |||
| 539286aafd | |||
| 23522b7b55 | |||
| bf3fac56e4 | |||
| a5bf8e9075 | |||
| 1d31b8f7e4 | |||
| b144c7dccc | |||
| 1364975396 | |||
| deaa7f50f8 | |||
| 744ab5156f | |||
| c45413969a | |||
| b314e5e080 | |||
| 17129e2eaa | |||
| 654fd8d74c | |||
| 9d3568ef75 | |||
| 14712cac88 | |||
| 0d568c758b | |||
| 7c6b88c7c5 | |||
| 32c93be46e | |||
| 7de8d85199 | |||
| f7dd65a3de | |||
| d8cdbe0041 | |||
| 2b8b6d3ea9 | |||
| 936c7e389f | |||
| 6864b4207b | |||
| 98eb5b54be | |||
| 3332e6e236 | |||
| 0533da72d7 | |||
| a1de238716 | |||
| f0d112254b | |||
| 830a7397ef | |||
| 5428765329 | |||
| 23c912f2b7 | |||
| 9c4b023297 | |||
| 53037b5ed8 | |||
| e2546a653d | |||
| 4b8cada873 | |||
| fa3ca1d08a | |||
| 8dd5cb9602 | |||
| a054f7be9c | |||
| df314dc6d1 | |||
| 930280f4ce | |||
| 5022c1ae29 | |||
| 6ced756a6b | |||
| b17268db50 | |||
| 476da37009 | |||
| 455f059c6f | |||
| 5255a37c93 | |||
| 68dc274f72 | |||
| 30228f7f8e | |||
| e15ef79ca9 | |||
| bc012a7518 | |||
| 3b4409cfad | |||
| d3726134b2 | |||
| 5acb7f1c55 | |||
| 81336668b3 | |||
| 35c2b83015 | |||
| cc1ee1deaa | |||
| 29bd038579 | |||
| f6c4f86986 | |||
| 68183e9dce | |||
| 78ec91a3a9 | |||
| c95d458e52 | |||
| 191ae3ec1e | |||
| ab9598d00a | |||
| 0f8a2e624a | |||
| d77e8da3f3 | |||
| a27eeb3255 | |||
| 413ccb83e6 | |||
| 797bb567c6 | |||
| f2a5dc40ee | |||
| 3979480532 | |||
| 76f1993e7a | |||
| d783fa2b89 | |||
| bbce18caac | |||
| 3ce2d8a656 | |||
| a5c86a2f5c | |||
| d18e533adf | |||
| 2b881aaad0 | |||
| 39cc07608f | |||
| 9894cfcced | |||
| b5d80be037 | |||
| b7870fbd9b | |||
| 36e6d486fc | |||
| 2d5dc84f1a | |||
| b47405e1bd | |||
| 8b64deab40 | |||
| c8846e0e93 | |||
| 7641cba01d | |||
| 4dc1785ef1 | |||
| b2286f3e34 | |||
| 65a20aa457 | |||
| d8a7d71344 | |||
| bb490df9a6 | |||
| cdfd6519c8 | |||
| e8a2846449 | |||
| d065cbf934 | |||
| 413b107b9a | |||
| c336292346 | |||
| adf50f1e81 | |||
| 636bc0a99d | |||
| a7a61fae1d | |||
| 5ec12212e4 | |||
| 77c90a308e | |||
| 4a8c50f886 | |||
| a86d7f52e9 | |||
| 4820ea15d6 | |||
| b5de605e2b | |||
| d6ed2050d4 | |||
| 16e123b7bb | |||
| 4eb91683a9 | |||
| 0cb78b9067 | |||
| e226a89637 | |||
| 431f8c2c6a | |||
| 19a9141c2d | |||
| bc649b9a85 | |||
| 702067e521 | |||
| 03a84daf9d | |||
| b91d922600 | |||
| ed02aebf9a | |||
| 1a048390fd | |||
| 1741d3bef6 | |||
| 540a0a3685 | |||
| d2fd3ce434 | |||
| ea76868d65 | |||
| f31351bedb | |||
| 8863983c7b | |||
| 64a34cac32 | |||
| 352e71461d | |||
| 87d0b5c76f | |||
| d0af018b8d | |||
| 55e9a1cbd6 | |||
| 65bafb75b1 | |||
| 01fd1c2437 | |||
| 78ba4468b0 | |||
| 8581c7ecce | |||
| 9be6fe6bc3 | |||
| 8c506da21e | |||
| c02002eb4b | |||
| 57ecfca862 | |||
| 28d41e9397 | |||
| 7a1866d280 | |||
| d229b108c3 | |||
| 39640cb697 | |||
| 9ecf2e9feb | |||
| 2db07cdb1f | |||
| faa29ef285 | |||
| 16b0d5b829 | |||
| 6ae33d04b8 | |||
| 333eb8d60f | |||
| 414c69fd62 | |||
| 9951b58005 | |||
| e73ed82ae9 | |||
| 8bde34a24f | |||
| b58380e505 | |||
| 16f8de810c | |||
| 70f2de4fd3 | |||
| a5d5e5825f | |||
| 0f16c72762 | |||
| 9ca7a0d6d1 | |||
| 7d5bfd8c9f | |||
| cc9a06b116 | |||
| b8fc7b0c9e | |||
| 3999f2a373 | |||
| 4eb2c0e123 | |||
| b8a838aee1 | |||
| 84e5932ea5 | |||
| e41573ca74 | |||
| 4c5c99e6ae | |||
| dcb940ba95 | |||
| 8d3b66f7e3 | |||
| bc89b6ea74 | |||
| f0742dffa2 | |||
| 6c71a1020d | |||
| 1db3e43adf | |||
| cb59b0b5e4 | |||
| 8e0f05055e | |||
| 4768bacf1c | |||
| dc206c0999 | |||
| 77e1983b2e | |||
| d344ee226c | |||
| 3d0e4141bf | |||
| 01fb216ff7 | |||
| a662b2a6c6 | |||
| b1af82eba8 | |||
| 378ef5246e | |||
| 606814f10e | |||
| 5e06a0d001 | |||
| c0e3274375 | |||
| 119ec5e405 | |||
| 79efa51941 | |||
| 701d0b21ef | |||
| 0f23d5f967 | |||
| 36b26e08c3 | |||
| ac08638a63 | |||
| 03146946fa | |||
| 0f9a10c598 | |||
| 2bd6881361 | |||
| ba208f5b48 | |||
| bdef85f7db | |||
| 2cb47938fd | |||
| dfe0b414ac | |||
| 1864f4cb38 | |||
| 7c39d9f0c1 | |||
| 79f5a1d052 | |||
| 6fed75bb45 | |||
| 352ed3b6a1 | |||
| b37691711c | |||
| 13fda2efe1 | |||
| 3c3d98b9c3 | |||
| f582d70031 | |||
| 1ac8aef4de | |||
| 4754372fcd | |||
| eac85779eb | |||
| b0d8711b65 | |||
| f0844ed923 | |||
| 129242534d | |||
| 6481b555b4 | |||
| 794e51494e | |||
| 3059e96041 | |||
| bd595f84e8 | |||
| 344e7470f6 |
@@ -1 +0,0 @@
|
||||
OPENAI_API_KEY=
|
||||
@@ -5,7 +5,7 @@ body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
#### Before submitting a bug, please make sure the issue hasn't been already addressed by searching through [the existing and past issues](https://github.com/gventuri/pandas-ai/issues?q=is%3Aissue+sort%3Acreated-desc+).
|
||||
#### Before submitting a bug, please make sure the issue hasn't been already addressed by searching through [the existing and past issues](https://github.com/embedchain/embedchain/issues?q=is%3Aissue+sort%3Acreated-desc+).
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: 🐛 Describe the bug
|
||||
|
||||
@@ -2,14 +2,13 @@ name: Publish Python 🐍 distributions 📦 to PyPI and TestPyPI
|
||||
|
||||
on:
|
||||
release:
|
||||
types: [published] # This will trigger the workflow when you create a new release
|
||||
types: [published]
|
||||
|
||||
jobs:
|
||||
build-n-publish:
|
||||
name: Build and publish Python 🐍 distributions 📦 to PyPI and TestPyPI
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
# IMPORTANT: this permission is mandatory for trusted publishing
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
@@ -23,18 +22,25 @@ jobs:
|
||||
run: |
|
||||
curl -sSL https://install.python-poetry.org | python3 -
|
||||
echo "$HOME/.local/bin" >> $GITHUB_PATH
|
||||
|
||||
|
||||
- name: Install dependencies
|
||||
run: poetry install
|
||||
|
||||
run: |
|
||||
cd embedchain
|
||||
poetry install
|
||||
|
||||
- name: Build a binary wheel and a source tarball
|
||||
run: poetry build
|
||||
run: |
|
||||
cd embedchain
|
||||
poetry build
|
||||
|
||||
- name: Publish distribution 📦 to Test PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
repository_url: https://test.pypi.org/legacy/
|
||||
packages_dir: embedchain/dist/
|
||||
|
||||
- name: Publish distribution 📦 to PyPI
|
||||
if: startsWith(github.ref, 'refs/tags')
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages_dir: embedchain/dist/
|
||||
@@ -3,7 +3,15 @@ name: ci
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'embedchain/**'
|
||||
- 'embedchain/tests/**'
|
||||
- 'embedchain/examples/**'
|
||||
pull_request:
|
||||
paths:
|
||||
- 'embedchain/embedchain/**'
|
||||
- 'embedchain/tests/**'
|
||||
- 'embedchain/examples/**'
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -19,10 +27,27 @@ jobs:
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- name: Install poetry
|
||||
run: pip install poetry==1.4.2
|
||||
uses: snok/install-poetry@v1
|
||||
with:
|
||||
version: 1.4.2
|
||||
virtualenvs-create: true
|
||||
virtualenvs-in-project: true
|
||||
- name: Load cached venv
|
||||
id: cached-poetry-dependencies
|
||||
uses: actions/cache@v2
|
||||
with:
|
||||
path: .venv
|
||||
key: venv-${{ runner.os }}-${{ hashFiles('**/poetry.lock') }}
|
||||
- name: Install dependencies
|
||||
run: poetry install --all-extras
|
||||
run: cd embedchain && make install_all
|
||||
if: steps.cached-poetry-dependencies.outputs.cache-hit != 'true'
|
||||
- name: Lint with ruff
|
||||
run: make ci_lint
|
||||
- name: Test with pytest
|
||||
run: make ci_test
|
||||
run: cd embedchain && make lint
|
||||
- name: Run tests and generate coverage report
|
||||
run: cd embedchain && make coverage
|
||||
- name: Upload coverage reports to Codecov
|
||||
uses: codecov/codecov-action@v3
|
||||
with:
|
||||
file: coverage.xml
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
|
||||
@@ -76,7 +76,6 @@ docs/_build/
|
||||
target/
|
||||
|
||||
# Jupyter Notebook
|
||||
.ipynb_checkpoints
|
||||
|
||||
# IPython
|
||||
profile_default/
|
||||
@@ -165,9 +164,23 @@ cython_debug/
|
||||
|
||||
# Database
|
||||
db
|
||||
test-db
|
||||
!embedchain/embedchain/core/db/
|
||||
|
||||
.vscode
|
||||
/poetry.lock
|
||||
.idea/
|
||||
|
||||
.DS_Store
|
||||
|
||||
notebooks/*.yaml
|
||||
.ipynb_checkpoints/
|
||||
|
||||
!configs/*.yaml
|
||||
|
||||
# cache db
|
||||
*.db
|
||||
|
||||
# local directories for testing
|
||||
eval/
|
||||
qdrant_storage/
|
||||
.crossnote
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
repos:
|
||||
- repo: https://github.com/psf/black
|
||||
rev: 23.3.0
|
||||
hooks:
|
||||
- id: black
|
||||
- repo: https://github.com/charliermarsh/ruff-pre-commit
|
||||
rev: 'v0.0.220'
|
||||
hooks:
|
||||
- id: ruff
|
||||
name: ruff
|
||||
# Respect `exclude` and `extend-exclude` settings.
|
||||
args: ["--force-exclude"]
|
||||
- repo: local
|
||||
hooks:
|
||||
- id: pytest-check
|
||||
name: pytest-check
|
||||
entry: poetry run pytest
|
||||
language: system
|
||||
pass_filenames: false
|
||||
always_run: true
|
||||
@@ -1,30 +1,32 @@
|
||||
.PHONY: format sort lint
|
||||
|
||||
# Variables
|
||||
PYTHON := python3
|
||||
PIP := $(PYTHON) -m pip
|
||||
PROJECT_NAME := embedchain
|
||||
RUFF_OPTIONS = --line-length 120
|
||||
ISORT_OPTIONS = --profile black
|
||||
|
||||
# Targets
|
||||
.PHONY: install format lint clean test ci_lint ci_test
|
||||
|
||||
install:
|
||||
$(PIP) install --upgrade pip
|
||||
$(PIP) install -e .[dev]
|
||||
# Default target
|
||||
all: format sort lint
|
||||
|
||||
# Format code with ruff
|
||||
format:
|
||||
$(PYTHON) -m black .
|
||||
$(PYTHON) -m isort .
|
||||
poetry run ruff check . --fix $(RUFF_OPTIONS)
|
||||
|
||||
# Sort imports with isort
|
||||
sort:
|
||||
poetry run isort . $(ISORT_OPTIONS)
|
||||
|
||||
# Lint code with ruff
|
||||
lint:
|
||||
$(PYTHON) -m ruff .
|
||||
poetry run ruff check . $(RUFF_OPTIONS)
|
||||
|
||||
docs:
|
||||
cd docs && mintlify dev
|
||||
|
||||
build:
|
||||
poetry build
|
||||
|
||||
publish:
|
||||
poetry publish
|
||||
|
||||
clean:
|
||||
rm -rf dist build *.egg-info
|
||||
|
||||
test:
|
||||
$(PYTHON) -m pytest
|
||||
|
||||
ci_lint:
|
||||
poetry run ruff .
|
||||
|
||||
ci_test:
|
||||
poetry run pytest
|
||||
poetry run rm -rf dist
|
||||
|
||||
@@ -1,91 +1,107 @@
|
||||
# embedchain
|
||||
<p align="center">
|
||||
<img src="docs/images/mem0-bg.png" width="500px" alt="Mem0 Logo">
|
||||
</p>
|
||||
|
||||
[](https://pypi.org/project/embedchain/)
|
||||
[](https://discord.gg/CUU9FPhRNt)
|
||||
[](https://twitter.com/embedchain)
|
||||
[](https://embedchain.substack.com/)
|
||||
[](https://colab.research.google.com/drive/138lMWhENGeEu7Q1-6lNbNTHGLZXBBz_B?usp=sharing)
|
||||
<p align="center">
|
||||
<a href="https://embedchain.ai/slack">
|
||||
<img src="https://img.shields.io/badge/slack-embedchain-brightgreen.svg?logo=slack" alt="Slack">
|
||||
</a>
|
||||
<a href="https://embedchain.ai/discord">
|
||||
<img src="https://dcbadge.vercel.app/api/server/6PzXDgEjG5?style=flat" alt="Discord">
|
||||
</a>
|
||||
<a href="https://twitter.com/mem0ai">
|
||||
<img src="https://img.shields.io/twitter/follow/mem0ai" alt="Twitter">
|
||||
</a>
|
||||
</p>
|
||||
|
||||
Embedchain is a framework to easily create LLM powered bots over any dataset. If you want a javascript version, check out [embedchain-js](https://github.com/embedchain/embedchainjs)
|
||||
# Mem0: The Memory Layer for Personalized AI
|
||||
|
||||
## 🤝 Schedule a 1-on-1 Session
|
||||
Mem0 provides a smart, self-improving memory layer for Large Language Models, enabling personalized AI experiences across applications.
|
||||
|
||||
Book a [1-on-1 Session](https://cal.com/taranjeetio/ec) with Taranjeet, the founder, to discuss any issues, provide feedback, or explore how we can improve Embedchain for you.
|
||||
> Note: The Mem0 repository now also includes the Embedchain project. We continue to maintain and support Embedchain ❤️. You can find the Embedchain codebase in the [embedchain](https://github.com/mem0ai/mem0/tree/main/embedchain) directory.
|
||||
## 🚀 Quick Start
|
||||
|
||||
## 🔧 Quick install
|
||||
### Installation
|
||||
|
||||
```bash
|
||||
pip install --upgrade embedchain
|
||||
pip install mem0ai
|
||||
```
|
||||
|
||||
## 🔍 Demo
|
||||
### Basic Usage
|
||||
|
||||
Try out embedchain in your browser:
|
||||
```python
|
||||
from mem0 import Memory
|
||||
|
||||
[](https://colab.research.google.com/drive/138lMWhENGeEu7Q1-6lNbNTHGLZXBBz_B?usp=sharing)
|
||||
# Initialize Mem0
|
||||
m = Memory()
|
||||
|
||||
# Store a memory from any unstructured text
|
||||
result = m.add("I am working on improving my tennis skills. Suggest some online courses.", user_id="alice", metadata={"category": "hobbies"})
|
||||
print(result)
|
||||
# Created memory: Improving her tennis skills. Looking for online suggestions.
|
||||
|
||||
# Retrieve memories
|
||||
all_memories = m.get_all()
|
||||
print(all_memories)
|
||||
|
||||
# Search memories
|
||||
related_memories = m.search(query="What are Alice's hobbies?", user_id="alice")
|
||||
print(related_memories)
|
||||
|
||||
# Update a memory
|
||||
result = m.update(memory_id="m1", data="Likes to play tennis on weekends")
|
||||
print(result)
|
||||
|
||||
# Get memory history
|
||||
history = m.history(memory_id="m1")
|
||||
print(history)
|
||||
```
|
||||
|
||||
## 🔑 Core Features
|
||||
|
||||
- **Multi-Level Memory**: User, Session, and AI Agent memory retention
|
||||
- **Adaptive Personalization**: Continuous improvement based on interactions
|
||||
- **Developer-Friendly API**: Simple integration into various applications
|
||||
- **Cross-Platform Consistency**: Uniform behavior across devices
|
||||
- **Managed Service**: Hassle-free hosted solution
|
||||
|
||||
## 📖 Documentation
|
||||
|
||||
The documentation for embedchain can be found at [docs.embedchain.ai](https://docs.embedchain.ai).
|
||||
For detailed usage instructions and API reference, visit our documentation at [docs.mem0.ai](https://docs.mem0.ai).
|
||||
|
||||
## 💻 Usage
|
||||
## 🔧 Advanced Usage
|
||||
|
||||
Embedchain empowers you to create chatbot models similar to ChatGPT, using your own evolving dataset.
|
||||
|
||||
### Data Types Supported
|
||||
|
||||
* Youtube video
|
||||
* PDF file
|
||||
* Web page
|
||||
* Sitemap
|
||||
* Doc file
|
||||
* Code documentation website loader
|
||||
* Notion
|
||||
|
||||
### Queries
|
||||
|
||||
For example, you can use Embedchain to create an Elon Musk bot using the following code:
|
||||
For production environments, you can use Qdrant as a vector store:
|
||||
|
||||
```python
|
||||
import os
|
||||
from embedchain import App
|
||||
from mem0 import Memory
|
||||
|
||||
# Create a bot instance
|
||||
os.environ["OPENAI_API_KEY"] = "YOUR API KEY"
|
||||
elon_bot = App()
|
||||
|
||||
# Embed online resources
|
||||
elon_bot.add("https://en.wikipedia.org/wiki/Elon_Musk")
|
||||
elon_bot.add("https://tesla.com/elon-musk")
|
||||
elon_bot.add("https://www.youtube.com/watch?v=MxZpaJK74Y4")
|
||||
|
||||
# Query the bot
|
||||
elon_bot.query("How many companies does Elon Musk run?")
|
||||
# Answer: Elon Musk runs four companies: Tesla, SpaceX, Neuralink, and The Boring Company
|
||||
```
|
||||
|
||||
## 🤝 Contributing
|
||||
|
||||
Contributions are welcome! Please check out the issues on the repository, and feel free to open a pull request.
|
||||
For more information, please see the [contributing guidelines](CONTRIBUTING.md).
|
||||
|
||||
For more reference, please go through [Development Guide](https://docs.embedchain.ai/contribution/dev) and [Documentation Guide](https://docs.embedchain.ai/contribution/docs).
|
||||
|
||||
<a href="https://github.com/embedchain/embedchain/graphs/contributors">
|
||||
<img src="https://contrib.rocks/image?repo=embedchain/embedchain" />
|
||||
</a>
|
||||
|
||||
## Citation
|
||||
|
||||
If you utilize this repository, please consider citing it with:
|
||||
|
||||
```
|
||||
@misc{embedchain,
|
||||
author = {Taranjeet Singh},
|
||||
title = {Embedchain: Framework to easily create LLM powered bots over any dataset},
|
||||
year = {2023},
|
||||
publisher = {GitHub},
|
||||
journal = {GitHub repository},
|
||||
howpublished = {\url{https://github.com/embedchain/embedchain}},
|
||||
config = {
|
||||
"vector_store": {
|
||||
"provider": "qdrant",
|
||||
"config": {
|
||||
"host": "localhost",
|
||||
"port": 6333,
|
||||
}
|
||||
},
|
||||
}
|
||||
|
||||
m = Memory.from_config(config)
|
||||
```
|
||||
|
||||
## 🗺️ Roadmap
|
||||
|
||||
- Integration with various LLM providers
|
||||
- Support for LLM frameworks
|
||||
- Integration with AI Agents frameworks
|
||||
- Customizable memory creation/update rules
|
||||
- Hosted platform support
|
||||
|
||||
## 🙋♂️ Support
|
||||
Join our Slack or Discord community for support and discussions.
|
||||
If you have any questions, feel free to reach out to us using one of the following methods:
|
||||
|
||||
- [Join our Discord](https://embedchain.ai/discord)
|
||||
- [Join our Slack](https://embedchain.ai/slack)
|
||||
- [Follow us on Twitter](https://twitter.com/mem0ai)
|
||||
- [Email us](mailto:founders@mem0.ai)
|
||||
|
||||
@@ -1,7 +1,14 @@
|
||||
# Contributing to embedchain docs
|
||||
# Mintlify Starter Kit
|
||||
|
||||
Click on `Use this template` to copy the Mintlify starter kit. The starter kit contains examples including
|
||||
|
||||
### 👩💻 Development
|
||||
- Guide pages
|
||||
- Navigation
|
||||
- Customizations
|
||||
- API Reference pages
|
||||
- Use of popular components
|
||||
|
||||
### Development
|
||||
|
||||
Install the [Mintlify CLI](https://www.npmjs.com/package/mintlify) to preview the documentation changes locally. To install, use the following command
|
||||
|
||||
@@ -15,9 +22,9 @@ Run the following command at the root of your documentation (where mint.json is)
|
||||
mintlify dev
|
||||
```
|
||||
|
||||
### 😎 Publishing Changes
|
||||
### Publishing Changes
|
||||
|
||||
Changes will be deployed to production automatically after your PR is merged to the main branch.
|
||||
Install our Github App to auto propagate changes from your repo to your deployment. Changes will be deployed to production automatically after pushing to the default branch. Find the link to install on your dashboard.
|
||||
|
||||
#### Troubleshooting
|
||||
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
<CardGroup cols={3}>
|
||||
<Card title="Talk to founders" icon="calendar" href="https://cal.com/taranjeetio/meet">
|
||||
Talk to founders
|
||||
</Card>
|
||||
<Card title="Slack" icon="slack" href="https://embedchain.ai/slack" color="#4A154B">
|
||||
Join our slack community
|
||||
</Card>
|
||||
<Card title="Discord" icon="discord" href="https://discord.gg/6PzXDgEjG5" color="#7289DA">
|
||||
Join our discord community
|
||||
</Card>
|
||||
</CardGroup>
|
||||
@@ -1,25 +0,0 @@
|
||||
---
|
||||
title: '➕ Adding Data'
|
||||
---
|
||||
|
||||
## Add Dataset
|
||||
|
||||
- This step assumes that you have already created an `app` instance by either using `App`, `OpenSourceApp` or `CustomApp`. We are calling our app instance as `naval_chat_bot` 🤖
|
||||
|
||||
- Now use `.add` method to add any dataset.
|
||||
|
||||
```python
|
||||
# naval_chat_bot = App() or
|
||||
# naval_chat_bot = OpenSourceApp()
|
||||
|
||||
# Embed Online Resources
|
||||
naval_chat_bot.add("https://www.youtube.com/watch?v=3qHkcs3kG44")
|
||||
naval_chat_bot.add("https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf")
|
||||
naval_chat_bot.add("https://nav.al/feedback")
|
||||
naval_chat_bot.add("https://nav.al/agi")
|
||||
|
||||
# Embed Local Resources
|
||||
naval_chat_bot.add(("Who is Naval Ravikant?", "Naval Ravikant is an Indian-American entrepreneur and investor."))
|
||||
```
|
||||
|
||||
The possible formats to add data can be found on the [Supported Data Formats](/advanced/data_types) page.
|
||||
@@ -1,128 +0,0 @@
|
||||
---
|
||||
title: '📱 App types'
|
||||
---
|
||||
|
||||
## App Types
|
||||
|
||||
We have three types of App.
|
||||
|
||||
### App
|
||||
|
||||
```python
|
||||
from embedchain import App
|
||||
app = App()
|
||||
```
|
||||
|
||||
- `App` uses OpenAI's model, so these are paid models. 💸 You will be charged for embedding model usage and LLM usage.
|
||||
- `App` uses OpenAI's embedding model to create embeddings for chunks and ChatGPT API as LLM to get answer given the relevant docs. Make sure that you have an OpenAI account and an API key. If you don't have an API key, you can create one by visiting [this link](https://platform.openai.com/account/api-keys).
|
||||
- `App` is opinionated. It uses the best embedding model and LLM on the market.
|
||||
- Once you have the API key, set it in an environment variable called `OPENAI_API_KEY`
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ["OPENAI_API_KEY"] = "sk-xxxx"
|
||||
```
|
||||
|
||||
### Llama2App
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
from embedchain import Llama2App
|
||||
|
||||
os.environ['REPLICATE_API_TOKEN'] = "REPLICATE API TOKEN"
|
||||
|
||||
zuck_bot = Llama2App()
|
||||
|
||||
# Embed your data
|
||||
zuck_bot.add("https://www.youtube.com/watch?v=Ff4fRgnuFgQ")
|
||||
zuck_bot.add("https://en.wikipedia.org/wiki/Mark_Zuckerberg")
|
||||
|
||||
# Nice, your bot is ready now. Start asking questions to your bot.
|
||||
zuck_bot.query("Who is Mark Zuckerberg?")
|
||||
# Answer: Mark Zuckerberg is an American internet entrepreneur and business magnate. He is the co-founder and CEO of Facebook. Born in 1984, he dropped out of Harvard University to focus on his social media platform, which has since grown to become one of the largest and most influential technology companies in the world.
|
||||
|
||||
# Enable web search for your bot
|
||||
zuck_bot.online = True # enable internet access for the bot
|
||||
zuck_bot.query("Who owns the new threads app and when it was founded?")
|
||||
# Answer: Based on the context provided, the new Threads app is owned by Meta, the parent company of Facebook, Instagram, and WhatsApp.
|
||||
```
|
||||
|
||||
- `Llama2App` uses Replicate's LLM model, so these are paid models. You can get the `REPLICATE_API_TOKEN` by registering on [their website](https://replicate.com/account).
|
||||
- `Llama2App` uses OpenAI's embedding model to create embeddings for chunks. Make sure that you have an OpenAI account and an API key. If you don't have an API key, you can create one by visiting [this link](https://platform.openai.com/account/api-keys).
|
||||
|
||||
|
||||
### OpenSourceApp
|
||||
|
||||
```python
|
||||
from embedchain import OpenSourceApp
|
||||
app = OpenSourceApp()
|
||||
```
|
||||
|
||||
- `OpenSourceApp` uses open source embedding and LLM model. It uses `all-MiniLM-L6-v2` from Sentence Transformers library as the embedding model and `gpt4all` as the LLM.
|
||||
- Here there is no need to setup any api keys. You just need to install embedchain package and these will get automatically installed. 📦
|
||||
- Once you have imported and instantiated the app, every functionality from here onwards is the same for either type of app. 📚
|
||||
- `OpenSourceApp` is opinionated. It uses the best open source embedding model and LLM on the market.
|
||||
- extra dependencies are required for this app type. Install them with `pip install embedchain[opensource]`.
|
||||
|
||||
### CustomApp
|
||||
|
||||
```python
|
||||
from embedchain import CustomApp
|
||||
from embedchain.config import CustomAppConfig
|
||||
from embedchain.models import Providers, EmbeddingFunctions
|
||||
|
||||
config = CustomAppConfig(embedding_fn=EmbeddingFunctions.OPENAI, provider=Providers.OPENAI)
|
||||
app = CustomApp(config)
|
||||
```
|
||||
|
||||
- `CustomApp` is not opinionated.
|
||||
- Configuration required. It's for advanced users who want to mix and match different embedding models and LLMs. Configuration required.
|
||||
- while it's doing that, it's still providing abstractions through `Providers`.
|
||||
- paid and free/open source providers included.
|
||||
- Once you have imported and instantiated the app, every functionality from here onwards is the same for either type of app. 📚
|
||||
- Following providers are available for an LLM
|
||||
- OPENAI
|
||||
- ANTHPROPIC
|
||||
- VERTEX_AI
|
||||
- GPT4ALL
|
||||
- AZURE_OPENAI
|
||||
- Following embedding functions are available for an embedding function
|
||||
- OPENAI
|
||||
- HUGGING_FACE
|
||||
- VERTEX_AI
|
||||
- GPT4ALL
|
||||
- AZURE_OPENAI
|
||||
|
||||
|
||||
### PersonApp
|
||||
|
||||
```python
|
||||
from embedchain import PersonApp
|
||||
naval_chat_bot = PersonApp("name_of_person_or_character") #Like "Yoda"
|
||||
```
|
||||
|
||||
- `PersonApp` uses OpenAI's model, so these are paid models. 💸 You will be charged for embedding model usage and LLM usage.
|
||||
- `PersonApp` uses OpenAI's embedding model to create embeddings for chunks and ChatGPT API as LLM to get answer given the relevant docs. Make sure that you have an OpenAI account and an API key. If you don't have an API key, you can create one by visiting [this link](https://platform.openai.com/account/api-keys).
|
||||
- Once you have the API key, set it in an environment variable called `OPENAI_API_KEY`
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ["OPENAI_API_KEY"] = "sk-xxxx"
|
||||
```
|
||||
|
||||
#### Compatibility with other apps
|
||||
|
||||
- If there is any other app instance in your script or app, you can change the import as
|
||||
|
||||
```python
|
||||
from embedchain import App as EmbedChainApp
|
||||
from embedchain import OpenSourceApp as EmbedChainOSApp
|
||||
from embedchain import PersonApp as EmbedChainPersonApp
|
||||
|
||||
# or
|
||||
|
||||
from embedchain import App as ECApp
|
||||
from embedchain import OpenSourceApp as ECOSApp
|
||||
from embedchain import PersonApp as ECPApp
|
||||
```
|
||||
@@ -1,89 +0,0 @@
|
||||
---
|
||||
title: '⚙️ Custom configurations'
|
||||
---
|
||||
|
||||
Embedchain is made to work out of the box. However, for advanced users we're also offering configuration options. All of these configuration options are optional and have sane defaults.
|
||||
|
||||
## Examples
|
||||
|
||||
### General
|
||||
|
||||
Here's the readme example with configuration options.
|
||||
|
||||
```python
|
||||
import os
|
||||
from embedchain import App
|
||||
from embedchain.config import AppConfig, AddConfig, QueryConfig, ChunkerConfig
|
||||
from chromadb.utils import embedding_functions
|
||||
|
||||
# Example: set the log level for debugging
|
||||
config = AppConfig(log_level="DEBUG")
|
||||
naval_chat_bot = App(config)
|
||||
|
||||
# Example: specify a custom collection name
|
||||
config = AppConfig(collection_name="naval_chat_bot")
|
||||
naval_chat_bot = App(config)
|
||||
|
||||
# Example: define your own chunker config for `youtube_video`
|
||||
chunker_config = ChunkerConfig(chunk_size=1000, chunk_overlap=100, length_function=len)
|
||||
naval_chat_bot.add("https://www.youtube.com/watch?v=3qHkcs3kG44", AddConfig(chunker=chunker_config))
|
||||
|
||||
add_config = AddConfig()
|
||||
naval_chat_bot.add("https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf", config=add_config)
|
||||
naval_chat_bot.add("https://nav.al/feedback", config=add_config)
|
||||
naval_chat_bot.add("https://nav.al/agi", config=add_config)
|
||||
|
||||
naval_chat_bot.add(("Who is Naval Ravikant?", "Naval Ravikant is an Indian-American entrepreneur and investor."), config=add_config)
|
||||
|
||||
query_config = QueryConfig()
|
||||
print(naval_chat_bot.query("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?", config=query_config))
|
||||
```
|
||||
|
||||
### Custom prompt template
|
||||
|
||||
Here's the example of using custom prompt template with `.query`
|
||||
|
||||
```python
|
||||
from embedchain.config import QueryConfig
|
||||
from embedchain.embedchain import App
|
||||
from string import Template
|
||||
import wikipedia
|
||||
|
||||
einstein_chat_bot = App()
|
||||
|
||||
# Embed Wikipedia page
|
||||
page = wikipedia.page("Albert Einstein")
|
||||
einstein_chat_bot.add(page.content)
|
||||
|
||||
# Example: use your own custom template with `$context` and `$query`
|
||||
einstein_chat_template = Template("""
|
||||
You are Albert Einstein, a German-born theoretical physicist,
|
||||
widely ranked among the greatest and most influential scientists of all time.
|
||||
|
||||
Use the following information about Albert Einstein to respond to
|
||||
the human's query acting as Albert Einstein.
|
||||
Context: $context
|
||||
|
||||
Keep the response brief. If you don't know the answer, just say that you don't know, don't try to make up an answer.
|
||||
|
||||
Human: $query
|
||||
Albert Einstein:""")
|
||||
query_config = QueryConfig(template=einstein_chat_template, system_prompt="You are Albert Einstein.")
|
||||
queries = [
|
||||
"Where did you complete your studies?",
|
||||
"Why did you win nobel prize?",
|
||||
"Why did you divorce your first wife?",
|
||||
]
|
||||
for query in queries:
|
||||
response = einstein_chat_bot.query(query, config=query_config)
|
||||
print("Query: ", query)
|
||||
print("Response: ", response)
|
||||
|
||||
# Output
|
||||
# Query: Where did you complete your studies?
|
||||
# Response: I completed my secondary education at the Argovian cantonal school in Aarau, Switzerland.
|
||||
# Query: Why did you win nobel prize?
|
||||
# Response: I won the Nobel Prize in Physics in 1921 for my services to Theoretical Physics, particularly for my discovery of the law of the photoelectric effect.
|
||||
# Query: Why did you divorce your first wife?
|
||||
# Response: We divorced due to living apart for five years.
|
||||
```
|
||||
@@ -1,141 +0,0 @@
|
||||
---
|
||||
title: '📋 Supported data formats'
|
||||
---
|
||||
|
||||
## Automatic data type detection
|
||||
The add method automatically tries to detect the data_type, based on your input for the source argument. So `app.add('https://www.youtube.com/watch?v=dQw4w9WgXcQ')` is enough to embed a YouTube video.
|
||||
|
||||
This detection is implemented for all formats. It is based on factors such as whether it's a URL, a local file, the source data type, etc.
|
||||
|
||||
### Debugging automatic detection
|
||||
|
||||
|
||||
Set `log_level=DEBUG` (in [AppConfig](http://localhost:3000/advanced/query_configuration#appconfig)) and make sure it's working as intended.
|
||||
|
||||
Otherwise, you will not know when, for instance, an invalid filepath is interpreted as raw text instead.
|
||||
|
||||
### Forcing a data type
|
||||
|
||||
To omit any issues with the data type detection, you can **force** a data_type by adding it as a `add` method argument.
|
||||
The examples below show you the keyword to force the respective `data_type`.
|
||||
|
||||
Forcing can also be used for edge cases, such as interpreting a sitemap as a web_page, for reading its raw text instead of following links.
|
||||
|
||||
## Remote Data Types
|
||||
|
||||
<Tip>
|
||||
**Use local files in remote data types**
|
||||
|
||||
Some data_types are meant for remote content and only work with URLs.
|
||||
You can pass local files by formatting the path using the `file:` [URI scheme](https://en.wikipedia.org/wiki/File_URI_scheme), e.g. `file:///info.pdf`.
|
||||
</Tip>
|
||||
|
||||
### Youtube video
|
||||
|
||||
To add any youtube video to your app, use the data_type (first argument to `.add()` method) as `youtube_video`. Eg:
|
||||
|
||||
```python
|
||||
app.add('a_valid_youtube_url_here', data_type='youtube_video')
|
||||
```
|
||||
|
||||
### PDF file
|
||||
|
||||
To add any pdf file, use the data_type as `pdf_file`. Eg:
|
||||
|
||||
```python
|
||||
app.add('a_valid_url_where_pdf_file_can_be_accessed', data_type='pdf_file')
|
||||
```
|
||||
|
||||
Note that we do not support password protected pdfs.
|
||||
|
||||
### Web page
|
||||
|
||||
To add any web page, use the data_type as `web_page`. Eg:
|
||||
|
||||
```python
|
||||
app.add('a_valid_web_page_url', data_type='web_page')
|
||||
```
|
||||
|
||||
### Sitemap
|
||||
|
||||
Add all web pages from an xml-sitemap. Filters non-text files. Use the data_type as `sitemap`. Eg:
|
||||
|
||||
```python
|
||||
app.add('https://example.com/sitemap.xml', data_type='sitemap')
|
||||
```
|
||||
|
||||
### Doc file
|
||||
|
||||
To add any doc/docx file, use the data_type as `docx`. `docx` allows remote urls and conventional file paths. Eg:
|
||||
|
||||
```python
|
||||
app.add('https://example.com/content/intro.docx', data_type="docx")
|
||||
app.add('content/intro.docx', data_type="docx")
|
||||
```
|
||||
|
||||
### Code documentation website loader
|
||||
|
||||
To add any code documentation website as a loader, use the data_type as `docs_site`. Eg:
|
||||
|
||||
```python
|
||||
app.add("https://docs.embedchain.ai/", data_type="docs_site")
|
||||
```
|
||||
|
||||
### Notion
|
||||
To use notion you must install the extra dependencies with `pip install embedchain[notion]`.
|
||||
|
||||
To load a notion page, use the data_type as `notion`. Since it is hard to automatically detect, forcing this is advised.
|
||||
The next argument must **end** with the `notion page id`. The id is a 32-character string. Eg:
|
||||
|
||||
```python
|
||||
app.add("cfbc134ca6464fc980d0391613959196", "notion")
|
||||
app.add("my-page-cfbc134ca6464fc980d0391613959196", "notion")
|
||||
app.add("https://www.notion.so/my-page-cfbc134ca6464fc980d0391613959196", "notion")
|
||||
```
|
||||
|
||||
## Local Data Types
|
||||
|
||||
### Text
|
||||
|
||||
To supply your own text, use the data_type as `text` and enter a string. The text is not processed, this can be very versatile. Eg:
|
||||
|
||||
```python
|
||||
app.add('Seek wealth, not money or status. Wealth is having assets that earn while you sleep. Money is how we transfer time and wealth. Status is your place in the social hierarchy.', data_type='text')
|
||||
```
|
||||
|
||||
Note: This is not used in the examples because in most cases you will supply a whole paragraph or file, which did not fit.
|
||||
|
||||
### QnA pair
|
||||
|
||||
To supply your own QnA pair, use the data_type as `qna_pair` and enter a tuple. Eg:
|
||||
|
||||
```python
|
||||
app.add(("Question", "Answer"), data_type="qna_pair")
|
||||
```
|
||||
|
||||
## Reusing a vector database
|
||||
|
||||
Default behavior is to create a persistent vector DB in the directory **./db**. You can split your application into two Python scripts: one to create a local vector DB and the other to reuse this local persistent vector DB. This is useful when you want to index hundreds of documents and separately implement a chat interface.
|
||||
|
||||
Create a local index:
|
||||
|
||||
```python
|
||||
from embedchain import App
|
||||
|
||||
naval_chat_bot = App()
|
||||
naval_chat_bot.add("https://www.youtube.com/watch?v=3qHkcs3kG44")
|
||||
naval_chat_bot.add("https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf")
|
||||
```
|
||||
|
||||
You can reuse the local index with the same code, but without adding new documents:
|
||||
|
||||
```python
|
||||
from embedchain import App
|
||||
|
||||
naval_chat_bot = App()
|
||||
print(naval_chat_bot.query("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?"))
|
||||
```
|
||||
|
||||
## More formats (coming soon!)
|
||||
|
||||
- If you want to add any other format, please create an [issue](https://github.com/embedchain/embedchain/issues) and we will add it to the list of supported formats.
|
||||
@@ -1,75 +0,0 @@
|
||||
---
|
||||
title: '🤝 Interface types'
|
||||
---
|
||||
|
||||
## Interface Types
|
||||
|
||||
The embedchain app exposes the following methods.
|
||||
|
||||
### Query Interface
|
||||
|
||||
- This interface is like a question answering bot. It takes a question and gets the answer. It does not maintain context about the previous chats.❓
|
||||
|
||||
- To use this, call `.query()` function to get the answer for any query.
|
||||
|
||||
```python
|
||||
print(naval_chat_bot.query("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?"))
|
||||
# answer: Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.
|
||||
```
|
||||
|
||||
### Chat Interface
|
||||
|
||||
- This interface is a chat interface that remembers previous conversations. Right now it remembers 5 conversations by default. 💬
|
||||
|
||||
- To use this, call `.chat` function to get the answer for any query.
|
||||
|
||||
```python
|
||||
print(naval_chat_bot.chat("How to be happy in life?"))
|
||||
# answer: The most important trick to being happy is to realize happiness is a skill you develop and a choice you make. You choose to be happy, and then you work at it. It's just like building muscles or succeeding at your job. It's about recognizing the abundance and gifts around you at all times.
|
||||
|
||||
print(naval_chat_bot.chat("who is naval ravikant?"))
|
||||
# answer: Naval Ravikant is an Indian-American entrepreneur and investor.
|
||||
|
||||
print(naval_chat_bot.chat("what did the author say about happiness?"))
|
||||
# answer: The author, Naval Ravikant, believes that happiness is a choice you make and a skill you develop. He compares the mind to the body, stating that just as the body can be molded and changed, so can the mind. He emphasizes the importance of being present in the moment and not getting caught up in regrets of the past or worries about the future. By being present and grateful for where you are, you can experience true happiness.
|
||||
```
|
||||
|
||||
#### Dry Run
|
||||
|
||||
Dry Run is an option in the `query` and `chat` methods that allows the user to not send their constructed prompt to the LLM, to save money. It's used for [testing](/advanced/testing#dry-run).
|
||||
|
||||
|
||||
### Stream Response
|
||||
|
||||
- You can add config to your query method to stream responses like ChatGPT does. You would require a downstream handler to render the chunk in your desirable format. Supports both OpenAI model and OpenSourceApp. 📊
|
||||
|
||||
- To use this, instantiate a `QueryConfig` or `ChatConfig` object with `stream=True`. Then pass it to the `.chat()` or `.query()` method. The following example iterates through the chunks and prints them as they appear.
|
||||
|
||||
```python
|
||||
app = App()
|
||||
query_config = QueryConfig(stream = True)
|
||||
resp = app.query("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?", query_config)
|
||||
|
||||
for chunk in resp:
|
||||
print(chunk, end="", flush=True)
|
||||
# answer: Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.
|
||||
```
|
||||
|
||||
### Other Methods
|
||||
|
||||
#### Reset
|
||||
|
||||
Resets the database and deletes all embeddings. Irreversible. Requires reinitialization afterwards.
|
||||
|
||||
```python
|
||||
app.reset()
|
||||
```
|
||||
|
||||
#### Count
|
||||
|
||||
Counts the number of embeddings (chunks) in the database.
|
||||
|
||||
```python
|
||||
print(app.count())
|
||||
# returns: 481
|
||||
```
|
||||
@@ -1,77 +0,0 @@
|
||||
---
|
||||
title: '🔍 Query configurations'
|
||||
---
|
||||
|
||||
## AppConfig
|
||||
|
||||
| option | description | type | default |
|
||||
|-----------|-----------------------|---------------------------------|------------------------|
|
||||
| log_level | log level | string | WARNING |
|
||||
| embedding_fn| embedding function | chromadb.utils.embedding_functions | \{text-embedding-ada-002\} |
|
||||
| db | vector database (experimental) | BaseVectorDB | ChromaDB |
|
||||
| collection_name | initial collection name for the database | string | embedchain_store |
|
||||
| collect_metrics | collect anonymous telemetry data to improve embedchain | boolean | true |
|
||||
|
||||
|
||||
## AddConfig
|
||||
|
||||
|option|description|type|default|
|
||||
|---|---|---|---|
|
||||
|chunker|chunker config|ChunkerConfig|Default values for chunker depends on the `data_type`. Please refer [ChunkerConfig](#chunker-config)|
|
||||
|loader|loader config|LoaderConfig|None|
|
||||
|
||||
Yes, you are passing `ChunkerConfig` to `AddConfig`, like so:
|
||||
|
||||
```python
|
||||
chunker_config = ChunkerConfig(chunk_size=100)
|
||||
add_config = AddConfig(chunker=chunker_config)
|
||||
app.add("lorem ipsum", config=add_config)
|
||||
```
|
||||
|
||||
### ChunkerConfig
|
||||
|
||||
|option|description|type|default|
|
||||
|---|---|---|---|
|
||||
|chunk_size|Maximum size of chunks to return|int|Default value for various `data_type` mentioned below|
|
||||
|chunk_overlap|Overlap in characters between chunks|int|Default value for various `data_type` mentioned below|
|
||||
|length_function|Function that measures the length of given chunks|typing.Callable|Default value for various `data_type` mentioned below|
|
||||
|
||||
Default values of chunker config parameters for different `data_type`:
|
||||
|
||||
|data_type|chunk_size|chunk_overlap|length_function|
|
||||
|---|---|---|---|
|
||||
|docx|1000|0|len|
|
||||
|text|300|0|len|
|
||||
|qna_pair|300|0|len|
|
||||
|web_page|500|0|len|
|
||||
|pdf_file|1000|0|len|
|
||||
|youtube_video|2000|0|len|
|
||||
|docs_site|500|50|len|
|
||||
|notion|300|0|len|
|
||||
|
||||
### LoaderConfig
|
||||
|
||||
_coming soon_
|
||||
|
||||
## QueryConfig
|
||||
|
||||
|option|description|type|default|
|
||||
|---|---|---|---|
|
||||
|number_documents|Absolute number of documents to pull from the database as context.|int|1
|
||||
|template|custom template for prompt. If history is used with query, $history has to be included as well.|Template|Template("Use the following pieces of context to answer the query at the end. If you don't know the answer, just say that you don't know, don't try to make up an answer. \$context Query: \$query Helpful Answer:")|
|
||||
|model|name of the model used.|string|depends on app type|
|
||||
|temperature|Controls the randomness of the model's output. Higher values (closer to 1) make output more random, lower values make it more deterministic.|float|0|
|
||||
|max_tokens|Controls how many tokens are used. Exact implementation (whether it counts prompt and/or response) depends on the model.|int|1000|
|
||||
|top_p|Controls the diversity of words. Higher values (closer to 1) make word selection more diverse, lower values make words less diverse.|float|1|
|
||||
|history|include conversation history from your client or database.|any (recommendation: list[str])|None|
|
||||
|stream|control if response is streamed back to the user.|bool|False|
|
||||
|deployment_name|t.b.a.|str|None|
|
||||
|system_prompt|System prompt string. Unused if none.|str|None|
|
||||
|
||||
## ChatConfig
|
||||
|
||||
All options for query and...
|
||||
|
||||
_coming soon_
|
||||
|
||||
`history` is not supported, as that is handled is handled automatically, the config option is not supported.
|
||||
@@ -1,29 +0,0 @@
|
||||
---
|
||||
title: '🧪 Testing'
|
||||
---
|
||||
|
||||
## Methods for testing
|
||||
|
||||
### Dry Run
|
||||
|
||||
Before you consume valueable tokens, you should make sure that the embedding you have done works and that it's receiving the correct document from the database.
|
||||
|
||||
For this you can use the `dry_run` option in your `query` or `chat` method.
|
||||
|
||||
Following the example above, add this to your script:
|
||||
|
||||
```python
|
||||
print(naval_chat_bot.query('Can you tell me who Naval Ravikant is?', dry_run=True))
|
||||
|
||||
'''
|
||||
Use the following pieces of context to answer the query at the end. If you don't know the answer, just say that you don't know, don't try to make up an answer.
|
||||
Q: Who is Naval Ravikant?
|
||||
A: Naval Ravikant is an Indian-American entrepreneur and investor.
|
||||
Query: Can you tell me who Naval Ravikant is?
|
||||
Helpful Answer:
|
||||
'''
|
||||
```
|
||||
|
||||
_The embedding is confirmed to work as expected. It returns the right document, even if the question is asked slightly different. No prompt tokens have been consumed._
|
||||
|
||||
**The dry run will still consume tokens to embed your query, but it is only ~1/15 of the prompt.**
|
||||
@@ -1,34 +0,0 @@
|
||||
---
|
||||
title: '💾 Vector Database'
|
||||
---
|
||||
|
||||
We support `Chroma` and `Elasticsearch` as two vector database.
|
||||
`Chroma` is used as a default database.
|
||||
|
||||
### Elasticsearch
|
||||
In order to use `Elasticsearch` as vector database we need to use App type `CustomApp`.
|
||||
```python
|
||||
import os
|
||||
from embedchain import CustomApp
|
||||
from embedchain.config import CustomAppConfig, ElasticsearchDBConfig
|
||||
from embedchain.models import Providers, EmbeddingFunctions, VectorDatabases
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = 'OPENAI_API_KEY'
|
||||
|
||||
es_config = ElasticsearchDBConfig(
|
||||
# elasticsearch url or list of nodes url with different hosts and ports.
|
||||
es_url='http://localhost:9200',
|
||||
# pass named parameters supported by Python Elasticsearch client
|
||||
ca_certs="/path/to/http_ca.crt",
|
||||
basic_auth=("username", "password")
|
||||
)
|
||||
config = CustomAppConfig(
|
||||
embedding_fn=EmbeddingFunctions.OPENAI,
|
||||
provider=Providers.OPENAI,
|
||||
db_type=VectorDatabases.ELASTICSEARCH,
|
||||
es_config=es_config,
|
||||
)
|
||||
es_app = CustomApp(config)
|
||||
```
|
||||
- Set `db_type=VectorDatabases.ELASTICSEARCH` and `es_config=ElasticsearchDBConfig(es_url='')` in `CustomAppConfig`.
|
||||
- `ElasticsearchDBConfig` accepts `es_url` as elasticsearch url or as list of nodes url with different hosts and ports. Additionally we can pass named parameters supported by Python Elasticsearch client.
|
||||
@@ -1,91 +0,0 @@
|
||||
---
|
||||
title: '🌍 API Server'
|
||||
---
|
||||
|
||||
The API Server based on Flask integrates the `embedchain` package, offering endpoints to add, query, and chat to engage in conversations with a chatbot using JSON requests.
|
||||
|
||||
### 🐳 Docker Setup
|
||||
|
||||
- Open variables.env, and edit it to add your 🔑 `OPENAI_API_KEY`.
|
||||
- To setup your api server using docker, run the following command inside this folder using your terminal.
|
||||
|
||||
```bash
|
||||
docker-compose up --build
|
||||
```
|
||||
|
||||
📝 Note: The build command might take a while to install all the packages depending on your system resources.
|
||||
|
||||
### 🚀 Usage Instructions
|
||||
|
||||
- Your api server is running on [http://localhost:5000/](http://localhost:5000/)
|
||||
- To use the api server, make an api call to the endpoints `/add`, `/query` and `/chat` using the json formats discussed below.
|
||||
- To add data sources to the bot (/add):
|
||||
```json
|
||||
// Request
|
||||
{
|
||||
"data_type": "your_data_type_here",
|
||||
"url_or_text": "your_url_or_text_here"
|
||||
}
|
||||
|
||||
// Response
|
||||
{
|
||||
"data": "Added data_type: url_or_text"
|
||||
}
|
||||
```
|
||||
- To ask queries from the bot (/query):
|
||||
```json
|
||||
// Request
|
||||
{
|
||||
"question": "your_question_here"
|
||||
}
|
||||
|
||||
// Response
|
||||
{
|
||||
"data": "your_answer_here"
|
||||
}
|
||||
```
|
||||
- To chat with the bot (/chat):
|
||||
```json
|
||||
// Request
|
||||
{
|
||||
"question": "your_question_here"
|
||||
}
|
||||
|
||||
// Response
|
||||
{
|
||||
"data": "your_answer_here"
|
||||
}
|
||||
```
|
||||
|
||||
### 📡 Curl Call Formats
|
||||
|
||||
- To add data sources to the bot (/add):
|
||||
```bash
|
||||
curl -X POST \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"data_type": "your_data_type_here",
|
||||
"url_or_text": "your_url_or_text_here"
|
||||
}' \
|
||||
http://localhost:5000/add
|
||||
```
|
||||
- To ask queries from the bot (/query):
|
||||
```bash
|
||||
curl -X POST \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"question": "your_question_here"
|
||||
}' \
|
||||
http://localhost:5000/query
|
||||
```
|
||||
- To chat with the bot (/chat):
|
||||
```bash
|
||||
curl -X POST \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"question": "your_question_here"
|
||||
}' \
|
||||
http://localhost:5000/chat
|
||||
```
|
||||
|
||||
🎉 Happy Chatting! 🎉
|
||||
@@ -0,0 +1,109 @@
|
||||
---
|
||||
title: Customer Support AI Agent
|
||||
---
|
||||
|
||||
You can create a personalized Customer Support AI Agent using Mem0. This guide will walk you through the necessary steps and provide the complete code to get you started.
|
||||
|
||||
## Overview
|
||||
|
||||
The Customer Support AI Agent leverages Mem0 to retain information across interactions, enabling a personalized and efficient support experience.
|
||||
|
||||
## Setup
|
||||
|
||||
Install the necessary packages using pip:
|
||||
|
||||
```bash
|
||||
pip install openai mem0ai
|
||||
```
|
||||
|
||||
## Full Code Example
|
||||
|
||||
Below is the simplified code to create and interact with a Customer Support AI Agent using Mem0:
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
from mem0 import Memory
|
||||
|
||||
# Set the OpenAI API key
|
||||
os.environ['OPENAI_API_KEY'] = 'sk-xxx'
|
||||
|
||||
class CustomerSupportAIAgent:
|
||||
def __init__(self):
|
||||
"""
|
||||
Initialize the CustomerSupportAIAgent with memory configuration and OpenAI client.
|
||||
"""
|
||||
config = {
|
||||
"vector_store": {
|
||||
"provider": "qdrant",
|
||||
"config": {
|
||||
"host": "localhost",
|
||||
"port": 6333,
|
||||
}
|
||||
},
|
||||
}
|
||||
self.memory = Memory.from_config(config)
|
||||
self.client = OpenAI()
|
||||
self.app_id = "customer-support"
|
||||
|
||||
def handle_query(self, query, user_id=None):
|
||||
"""
|
||||
Handle a customer query and store the relevant information in memory.
|
||||
|
||||
:param query: The customer query to handle.
|
||||
:param user_id: Optional user ID to associate with the memory.
|
||||
"""
|
||||
# Start a streaming chat completion request to the AI
|
||||
stream = self.client.chat.completions.create(
|
||||
model="gpt-4",
|
||||
stream=True,
|
||||
messages=[
|
||||
{"role": "system", "content": "You are a customer support AI agent."},
|
||||
{"role": "user", "content": query}
|
||||
]
|
||||
)
|
||||
# Store the query in memory
|
||||
self.memory.add(query, user_id=user_id, metadata={"app_id": self.app_id})
|
||||
|
||||
# Print the response from the AI in real-time
|
||||
for chunk in stream:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
|
||||
def get_memories(self, user_id=None):
|
||||
"""
|
||||
Retrieve all memories associated with the given customer ID.
|
||||
|
||||
:param user_id: Optional user ID to filter memories.
|
||||
:return: List of memories.
|
||||
"""
|
||||
return self.memory.get_all(user_id=user_id)
|
||||
|
||||
# Instantiate the CustomerSupportAIAgent
|
||||
support_agent = CustomerSupportAIAgent()
|
||||
|
||||
# Define a customer ID
|
||||
customer_id = "jane_doe"
|
||||
|
||||
# Handle a customer query
|
||||
support_agent.handle_query("I need help with my recent order. It hasn't arrived yet.", user_id=customer_id)
|
||||
```
|
||||
|
||||
### Fetching Memories
|
||||
|
||||
You can fetch all the memories at any point in time using the following code:
|
||||
|
||||
```python
|
||||
memories = support_agent.get_memories(user_id=customer_id)
|
||||
for m in memories:
|
||||
print(m['text'])
|
||||
```
|
||||
|
||||
### Key Points
|
||||
|
||||
- **Initialization**: The CustomerSupportAIAgent class is initialized with the necessary memory configuration and OpenAI client setup.
|
||||
- **Handling Queries**: The handle_query method sends a query to the AI and stores the relevant information in memory.
|
||||
- **Retrieving Memories**: The get_memories method fetches all stored memories associated with a customer.
|
||||
|
||||
### Conclusion
|
||||
|
||||
As the conversation progresses, Mem0's memory automatically updates based on the interactions, providing a continuously improving personalized support experience.
|
||||
@@ -1,22 +0,0 @@
|
||||
---
|
||||
title: '🌐 Full Stack'
|
||||
---
|
||||
|
||||
### 🐳 Docker Setup
|
||||
|
||||
- To setup full stack app using docker, run the following command inside this folder using your terminal.
|
||||
|
||||
```bash
|
||||
docker-compose up --build
|
||||
```
|
||||
|
||||
📝 Note: The build command might take a while to install all the packages depending on your system resources.
|
||||
|
||||
### 🚀 Usage Instructions
|
||||
|
||||
- Go to [http://localhost:3000/](http://localhost:3000/) in your browser to view the dashboard.
|
||||
- Add your `OpenAI API key` 🔑 in the Settings.
|
||||
- Create a new bot and you'll be navigated to its page.
|
||||
- Here you can add your data sources and then chat with the bot.
|
||||
|
||||
🎉 Happy Chatting! 🎉
|
||||
@@ -0,0 +1,32 @@
|
||||
---
|
||||
title: Overview
|
||||
description: How to use mem0 in your existing applications?
|
||||
---
|
||||
|
||||
|
||||
With Mem0, you can create stateful LLM-based applications such as chatbots, virtual assistants, or AI agents. Mem0 enhances your applications by providing a memory layer that makes responses:
|
||||
|
||||
- More personalized
|
||||
- More reliable
|
||||
- Cost-effective by reducing the number of LLM interactions
|
||||
- More engaging
|
||||
- Enables long-term memory
|
||||
|
||||
Here are some examples of how Mem0 can be integrated into various applications:
|
||||
|
||||
## Example Use Cases
|
||||
|
||||
<CardGroup cols={1}>
|
||||
<Card title="Personal AI Tutor" icon="square-1" href="/examples/personal-ai-tutor">
|
||||
<img width="100%" src="/images/ai-tutor.png" />
|
||||
Create a Personalized AI Tutor that adapts to student progress and learning preferences.
|
||||
</Card>
|
||||
<Card title="Personal Travel Assistant" icon="square-2" href="/examples/personal-travel-assistant">
|
||||
<img src="/images/personal-travel-agent.png" />
|
||||
Build a Personalized AI Travel Assistant that understands your travel preferences and past itineraries.
|
||||
</Card>
|
||||
<Card title="Customer Support Agent" icon="square-3" href="/examples/customer-support-agent">
|
||||
<img width="100%" src="/images/customer-support-agent.png" />
|
||||
Develop a Personal AI Assistant that remembers user preferences, past interactions, and context to provide personalized and efficient assistance.
|
||||
</Card>
|
||||
</CardGroup>
|
||||
@@ -0,0 +1,111 @@
|
||||
---
|
||||
title: Personalized AI Tutor
|
||||
---
|
||||
|
||||
You can create a personalized AI Tutor using Mem0. This guide will walk you through the necessary steps and provide the complete code to get you started.
|
||||
|
||||
## Overview
|
||||
|
||||
The Personalized AI Tutor leverages Mem0 to retain information across interactions, enabling a tailored learning experience. By integrating with OpenAI's GPT-4 model, the tutor can provide detailed and context-aware responses to user queries.
|
||||
|
||||
## Setup
|
||||
Before you begin, ensure you have the required dependencies installed. You can install the necessary packages using pip:
|
||||
|
||||
```bash
|
||||
pip install openai mem0ai
|
||||
```
|
||||
|
||||
## Full Code Example
|
||||
|
||||
Below is the complete code to create and interact with a Personalized AI Tutor using Mem0:
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
from mem0 import Memory
|
||||
|
||||
# Set the OpenAI API key
|
||||
os.environ['OPENAI_API_KEY'] = 'sk-xxx'
|
||||
|
||||
# Initialize the OpenAI client
|
||||
client = OpenAI()
|
||||
|
||||
class PersonalAITutor:
|
||||
def __init__(self):
|
||||
"""
|
||||
Initialize the PersonalAITutor with memory configuration and OpenAI client.
|
||||
"""
|
||||
config = {
|
||||
"vector_store": {
|
||||
"provider": "qdrant",
|
||||
"config": {
|
||||
"host": "localhost",
|
||||
"port": 6333,
|
||||
}
|
||||
},
|
||||
}
|
||||
self.memory = Memory.from_config(config)
|
||||
self.client = client
|
||||
self.app_id = "app-1"
|
||||
|
||||
def ask(self, question, user_id=None):
|
||||
"""
|
||||
Ask a question to the AI and store the relevant facts in memory
|
||||
|
||||
:param question: The question to ask the AI.
|
||||
:param user_id: Optional user ID to associate with the memory.
|
||||
"""
|
||||
# Start a streaming chat completion request to the AI
|
||||
stream = self.client.chat.completions.create(
|
||||
model="gpt-4",
|
||||
stream=True,
|
||||
messages=[
|
||||
{"role": "system", "content": "You are a personal AI Tutor."},
|
||||
{"role": "user", "content": question}
|
||||
]
|
||||
)
|
||||
# Store the question in memory
|
||||
self.memory.add(question, user_id=user_id, metadata={"app_id": self.app_id})
|
||||
|
||||
# Print the response from the AI in real-time
|
||||
for chunk in stream:
|
||||
if chunk.choices[0].delta.content is not None:
|
||||
print(chunk.choices[0].delta.content, end="")
|
||||
|
||||
def get_memories(self, user_id=None):
|
||||
"""
|
||||
Retrieve all memories associated with the given user ID.
|
||||
|
||||
:param user_id: Optional user ID to filter memories.
|
||||
:return: List of memories.
|
||||
"""
|
||||
return self.memory.get_all(user_id=user_id)
|
||||
|
||||
# Instantiate the PersonalAITutor
|
||||
ai_tutor = PersonalAITutor()
|
||||
|
||||
# Define a user ID
|
||||
user_id = "john_doe"
|
||||
|
||||
# Ask a question
|
||||
ai_tutor.ask("I am learning introduction to CS. What is queue? Briefly explain.", user_id=user_id)
|
||||
```
|
||||
|
||||
### Fetching Memories
|
||||
|
||||
You can fetch all the memories at any point in time using the following code:
|
||||
|
||||
```python
|
||||
memories = ai_tutor.get_memories(user_id=user_id)
|
||||
for m in memories:
|
||||
print(m['text'])
|
||||
```
|
||||
|
||||
### Key Points
|
||||
|
||||
- **Initialization**: The PersonalAITutor class is initialized with the necessary memory configuration and OpenAI client setup.
|
||||
- **Asking Questions**: The ask method sends a question to the AI and stores the relevant information in memory.
|
||||
- **Retrieving Memories**: The get_memories method fetches all stored memories associated with a user.
|
||||
|
||||
### Conclusion
|
||||
|
||||
As the conversation progresses, Mem0's memory automatically updates based on the interactions, providing a continuously improving personalized learning experience. This setup ensures that the AI Tutor can offer contextually relevant and accurate responses, enhancing the overall educational process.
|
||||
@@ -0,0 +1,101 @@
|
||||
---
|
||||
title: Personal AI Travel Assistant
|
||||
---
|
||||
Create a personalized AI Travel Assistant using Mem0. This guide provides step-by-step instructions and the complete code to get you started.
|
||||
|
||||
## Overview
|
||||
|
||||
The Personalized AI Travel Assistant uses Mem0 to store and retrieve information across interactions, enabling a tailored travel planning experience. It integrates with OpenAI's GPT-4 model to provide detailed and context-aware responses to user queries.
|
||||
|
||||
## Setup
|
||||
|
||||
Install the required dependencies using pip:
|
||||
|
||||
```bash
|
||||
pip install openai mem0ai
|
||||
```
|
||||
|
||||
## Full Code Example
|
||||
|
||||
Here's the complete code to create and interact with a Personalized AI Travel Assistant using Mem0:
|
||||
|
||||
```python
|
||||
import os
|
||||
from openai import OpenAI
|
||||
from mem0 import Memory
|
||||
|
||||
# Set the OpenAI API key
|
||||
os.environ['OPENAI_API_KEY'] = 'sk-xxx'
|
||||
|
||||
class PersonalTravelAssistant:
|
||||
def __init__(self):
|
||||
self.client = OpenAI()
|
||||
self.memory = Memory()
|
||||
self.messages = [{"role": "system", "content": "You are a personal AI Assistant."}]
|
||||
|
||||
def ask_question(self, question, user_id):
|
||||
# Fetch previous related memories
|
||||
previous_memories = self.search_memories(question, user_id=user_id)
|
||||
prompt = question
|
||||
if previous_memories:
|
||||
prompt = f"User input: {question}\n Previous memories: {previous_memories}"
|
||||
self.messages.append({"role": "user", "content": prompt})
|
||||
|
||||
# Generate response using GPT-4o
|
||||
response = self.client.chat.completions.create(
|
||||
model="gpt-4o",
|
||||
messages=self.messages
|
||||
)
|
||||
answer = response.choices[0].message.content
|
||||
self.messages.append({"role": "assistant", "content": answer})
|
||||
|
||||
# Store the question in memory
|
||||
self.memory.add(question, user_id=user_id)
|
||||
return answer
|
||||
|
||||
def get_memories(self, user_id):
|
||||
memories = self.memory.get_all(user_id=user_id)
|
||||
return [m['text'] for m in memories]
|
||||
|
||||
def search_memories(self, query, user_id):
|
||||
memories = self.memory.search(query, user_id=user_id)
|
||||
return [m['text'] for m in memories]
|
||||
|
||||
# Usage example
|
||||
user_id = "traveler_123"
|
||||
ai_assistant = PersonalTravelAssistant()
|
||||
|
||||
def main():
|
||||
while True:
|
||||
question = input("Question: ")
|
||||
if question.lower() in ['q', 'exit']:
|
||||
print("Exiting...")
|
||||
break
|
||||
|
||||
answer = ai_assistant.ask_question(question, user_id=user_id)
|
||||
print(f"Answer: {answer}")
|
||||
memories = ai_assistant.get_memories(user_id=user_id)
|
||||
print("Memories:")
|
||||
for memory in memories:
|
||||
print(f"- {memory}")
|
||||
print("-----")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
```
|
||||
|
||||
## Key Components
|
||||
|
||||
- **Initialization**: The `PersonalTravelAssistant` class is initialized with the OpenAI client and Mem0 memory setup.
|
||||
- **Asking Questions**: The `ask_question` method sends a question to the AI, incorporates previous memories, and stores new information.
|
||||
- **Memory Management**: The `get_memories` and search_memories methods handle retrieval and searching of stored memories.
|
||||
|
||||
## Usage
|
||||
|
||||
1. Set your OpenAI API key in the environment variable.
|
||||
2. Instantiate the `PersonalTravelAssistant`.
|
||||
3. Use the `main()` function to interact with the assistant in a loop.
|
||||
|
||||
## Conclusion
|
||||
|
||||
This Personalized AI Travel Assistant leverages Mem0's memory capabilities to provide context-aware responses. As you interact with it, the assistant learns and improves, offering increasingly personalized travel advice and information.
|
||||
@@ -1,39 +0,0 @@
|
||||
---
|
||||
title: '💼 Slack Bot'
|
||||
---
|
||||
|
||||
### 🖼️ Template Setup
|
||||
|
||||
- Fork [this](https://replit.com/@taranjeetio/EC-Slack-Bot-Template?v=1#README.md) replit template.
|
||||
- Set your `OPENAI_API_KEY` in Secrets.
|
||||
- Create a workspace on Slack if you don't have one already by clicking [here](https://slack.com/intl/en-in/).
|
||||
- Create a new App on your Slack account by going [here](https://api.slack.com/apps).
|
||||
- Select `From Scratch`, then enter the Bot Name and select your workspace.
|
||||
- On the `Basic Information` page copy the `Signing Secret` and set it in your secrets as `SLACK_SIGNING_SECRET`.
|
||||
- On the left Sidebar, go to `OAuth and Permissions` and add the following scopes under `Bot Token Scopes`:
|
||||
```text
|
||||
app_mentions:read
|
||||
channels:history
|
||||
channels:read
|
||||
chat:write
|
||||
```
|
||||
- Now select the option `Install to Workspace` and after it's done, copy the `Bot User OAuth Token` and set it in your secrets as `SLACK_BOT_TOKEN`.
|
||||
- Start your replit container now by clicking on `Run`.
|
||||
- On the Slack API website go to `Event Subscriptions` on the left Sidebar and turn on `Enable Events`.
|
||||
- Copy the generated server URL in replit, append `/chat` at its end and paste it in `Request URL` box.
|
||||
- After it gets verified, click on `Subscribe to bot events`, add `message.channels` Bot User Event and click on `Save Changes`.
|
||||
- Now go to your workspace, click on the bot name in the Sidebar and then add the bot to any channel you want.
|
||||
|
||||
### 🚀 Usage Instructions
|
||||
|
||||
- Go to the channel where you have added your bot.
|
||||
- To add data sources to the bot, use the command:
|
||||
```text
|
||||
add <data_type> <url_or_text>
|
||||
```
|
||||
- To ask queries from the bot, use the command:
|
||||
```text
|
||||
query <question>
|
||||
```
|
||||
|
||||
🎉 Happy Chatting! 🎉
|
||||
|
Before Width: | Height: | Size: 70 KiB |
@@ -0,0 +1,49 @@
|
||||
<svg width="24" height="24" viewBox="0 0 24 24" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M7.95343 21.1394C4.89586 21.1304 2.25471 19.458 0.987296 16.2895C-0.280118 13.121 0.108924 9.16314 1.74363 5.61505C4.8012 5.62409 7.44235 7.29648 8.70976 10.465C9.97718 13.6335 9.58814 17.5914 7.95343 21.1394Z" fill="white"/>
|
||||
<path d="M7.95343 21.1394C4.89586 21.1304 2.25471 19.458 0.987296 16.2895C-0.280118 13.121 0.108924 9.16314 1.74363 5.61505C4.8012 5.62409 7.44235 7.29648 8.70976 10.465C9.97718 13.6335 9.58814 17.5914 7.95343 21.1394Z" fill="url(#paint0_radial_101_2703)"/>
|
||||
<path d="M7.95343 21.1394C4.89586 21.1304 2.25471 19.458 0.987296 16.2895C-0.280118 13.121 0.108924 9.16314 1.74363 5.61505C4.8012 5.62409 7.44235 7.29648 8.70976 10.465C9.97718 13.6335 9.58814 17.5914 7.95343 21.1394Z" fill="black" fill-opacity="0.5" style="mix-blend-mode:hard-light"/>
|
||||
<path d="M7.95343 21.1394C4.89586 21.1304 2.25471 19.458 0.987296 16.2895C-0.280118 13.121 0.108924 9.16314 1.74363 5.61505C4.8012 5.62409 7.44235 7.29648 8.70976 10.465C9.97718 13.6335 9.58814 17.5914 7.95343 21.1394Z" fill="url(#paint1_linear_101_2703)" fill-opacity="0.5" style="mix-blend-mode:hard-light"/>
|
||||
<path d="M8.68359 10.4755C9.94543 13.63 9.56145 17.5723 7.9354 21.1112C4.89702 21.0957 2.27411 19.4306 1.01347 16.279C-0.248375 13.1245 0.135612 9.18218 1.76165 5.64328C4.80004 5.65883 7.42295 7.32386 8.68359 10.4755Z" stroke="url(#paint2_linear_101_2703)" stroke-opacity="0.05" stroke-width="0.056338"/>
|
||||
<path d="M7.31038 21.2574C11.3543 20.2215 14.8836 17.3754 16.6285 13.2361C18.3735 9.09671 17.9448 4.58749 15.8598 0.976291C11.8159 2.01214 8.2866 4.85826 6.54167 8.99762C4.79674 13.137 5.2254 17.6462 7.31038 21.2574Z" fill="white"/>
|
||||
<path d="M7.31038 21.2574C11.3543 20.2215 14.8836 17.3754 16.6285 13.2361C18.3735 9.09671 17.9448 4.58749 15.8598 0.976291C11.8159 2.01214 8.2866 4.85826 6.54167 8.99762C4.79674 13.137 5.2254 17.6462 7.31038 21.2574Z" fill="url(#paint3_radial_101_2703)"/>
|
||||
<path d="M16.6026 13.2251C14.8642 17.349 11.3512 20.1866 7.32411 21.2248C5.25257 17.624 4.82926 13.1324 6.56764 9.00855C8.30603 4.88472 11.819 2.04706 15.8461 1.00889C17.9176 4.60967 18.3409 9.10131 16.6026 13.2251Z" stroke="url(#paint4_linear_101_2703)" stroke-opacity="0.05" stroke-width="0.056338"/>
|
||||
<path d="M7.23368 21.2069C9.78906 23.2373 13.2102 23.9506 16.5772 22.8141C19.9441 21.6775 22.5058 18.9445 23.7304 15.6382C21.175 13.6078 17.7538 12.8944 14.3869 14.031C11.0199 15.1676 8.45822 17.9006 7.23368 21.2069Z" fill="white"/>
|
||||
<path d="M7.23368 21.2069C9.78906 23.2373 13.2102 23.9506 16.5772 22.8141C19.9441 21.6775 22.5058 18.9445 23.7304 15.6382C21.175 13.6078 17.7538 12.8944 14.3869 14.031C11.0199 15.1676 8.45822 17.9006 7.23368 21.2069Z" fill="url(#paint5_radial_101_2703)"/>
|
||||
<path d="M7.23368 21.2069C9.78906 23.2373 13.2102 23.9506 16.5772 22.8141C19.9441 21.6775 22.5058 18.9445 23.7304 15.6382C21.175 13.6078 17.7538 12.8944 14.3869 14.031C11.0199 15.1676 8.45822 17.9006 7.23368 21.2069Z" fill="black" fill-opacity="0.2" style="mix-blend-mode:hard-light"/>
|
||||
<path d="M7.23368 21.2069C9.78906 23.2373 13.2102 23.9506 16.5772 22.8141C19.9441 21.6775 22.5058 18.9445 23.7304 15.6382C21.175 13.6078 17.7538 12.8944 14.3869 14.031C11.0199 15.1676 8.45822 17.9006 7.23368 21.2069Z" fill="url(#paint6_linear_101_2703)" fill-opacity="0.5" style="mix-blend-mode:hard-light"/>
|
||||
<path d="M16.5682 22.7874C13.2176 23.9184 9.81361 23.2124 7.2672 21.1975C8.49194 17.9068 11.0444 15.189 14.3959 14.0577C17.7465 12.9266 21.1504 13.6326 23.6968 15.6476C22.4721 18.9383 19.9196 21.656 16.5682 22.7874Z" stroke="url(#paint7_linear_101_2703)" stroke-opacity="0.05" stroke-width="0.056338"/>
|
||||
<defs>
|
||||
<radialGradient id="paint0_radial_101_2703" cx="0" cy="0" r="1" gradientUnits="userSpaceOnUse" gradientTransform="translate(-3.00503 15.023) rotate(-10.029) scale(17.9572 17.784)">
|
||||
<stop stop-color="#00B0BB"/>
|
||||
<stop offset="1" stop-color="#00DB65"/>
|
||||
</radialGradient>
|
||||
<linearGradient id="paint1_linear_101_2703" x1="7.39036" y1="4.81308" x2="1.62975" y2="18.6894" gradientUnits="userSpaceOnUse">
|
||||
<stop stop-color="#18E299"/>
|
||||
<stop offset="1"/>
|
||||
</linearGradient>
|
||||
<linearGradient id="paint2_linear_101_2703" x1="7.94816" y1="8.01563" x2="1.7612" y2="18.746" gradientUnits="userSpaceOnUse">
|
||||
<stop/>
|
||||
<stop offset="1" stop-opacity="0"/>
|
||||
</linearGradient>
|
||||
<radialGradient id="paint3_radial_101_2703" cx="0" cy="0" r="1" gradientUnits="userSpaceOnUse" gradientTransform="translate(8.11404 20.8822) rotate(-75.7542) scale(21.6246 23.7772)">
|
||||
<stop stop-color="#00BBBB"/>
|
||||
<stop offset="0.712616" stop-color="#00DB65"/>
|
||||
</radialGradient>
|
||||
<linearGradient id="paint4_linear_101_2703" x1="7.60205" y1="5.8709" x2="15.5561" y2="16.3719" gradientUnits="userSpaceOnUse">
|
||||
<stop/>
|
||||
<stop offset="1" stop-opacity="0"/>
|
||||
</linearGradient>
|
||||
<radialGradient id="paint5_radial_101_2703" cx="0" cy="0" r="1" gradientUnits="userSpaceOnUse" gradientTransform="translate(7.84537 21.5181) rotate(-20.3525) scale(18.5603 17.32)">
|
||||
<stop stop-color="#00B0BB"/>
|
||||
<stop offset="1" stop-color="#00DB65"/>
|
||||
</radialGradient>
|
||||
<linearGradient id="paint6_linear_101_2703" x1="16.8078" y1="13.0071" x2="10.0409" y2="22.9937" gradientUnits="userSpaceOnUse">
|
||||
<stop stop-color="#00B1BC"/>
|
||||
<stop offset="1"/>
|
||||
</linearGradient>
|
||||
<linearGradient id="paint7_linear_101_2703" x1="16.8078" y1="13.0071" x2="14.1687" y2="23.841" gradientUnits="userSpaceOnUse">
|
||||
<stop/>
|
||||
<stop offset="1" stop-opacity="0"/>
|
||||
</linearGradient>
|
||||
</defs>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 5.3 KiB |
|
After Width: | Height: | Size: 2.8 MiB |
|
After Width: | Height: | Size: 843 KiB |
|
Before Width: | Height: | Size: 256 KiB |
|
After Width: | Height: | Size: 4.6 MiB |
|
After Width: | Height: | Size: 13 KiB |
|
After Width: | Height: | Size: 13 KiB |
|
After Width: | Height: | Size: 565 KiB |
|
After Width: | Height: | Size: 3.9 MiB |
|
After Width: | Height: | Size: 180 KiB |
|
After Width: | Height: | Size: 169 KiB |
@@ -0,0 +1,85 @@
|
||||
---
|
||||
title: MultiOn
|
||||
---
|
||||
|
||||
Build personal browser agent remembers user preferences and automates web tasks. It integrates Mem0 for memory management with MultiOn for executing browser actions, enabling personalized and efficient web interactions.
|
||||
|
||||
## Overview
|
||||
|
||||
In this example, we will create a Browser based AI Agent that searches [arxiv.org](https://arxiv.org) for research papers relevant to user's research interests.
|
||||
|
||||
## Setup and Configuration
|
||||
|
||||
Install necessary libraries:
|
||||
|
||||
```bash
|
||||
pip install mem0ai multion
|
||||
```
|
||||
|
||||
First, we'll import the necessary libraries and set up our configurations.
|
||||
|
||||
```python
|
||||
import os
|
||||
from mem0 import Memory
|
||||
from multion.client import MultiOn
|
||||
|
||||
# Configuration
|
||||
OPENAI_API_KEY = 'sk-xxx' # Replace with your actual OpenAI API key
|
||||
MULTION_API_KEY = 'your-multion-key' # Replace with your actual MultiOn API key
|
||||
USER_ID = "deshraj"
|
||||
|
||||
# Set up OpenAI API key
|
||||
os.environ['OPENAI_API_KEY'] = OPENAI_API_KEY
|
||||
|
||||
# Initialize Mem0 and MultiOn
|
||||
memory = Memory()
|
||||
multion = MultiOn(api_key=MULTION_API_KEY)
|
||||
```
|
||||
|
||||
## Add memories to Mem0
|
||||
|
||||
Next, we'll define our user data and add it to Mem0.
|
||||
|
||||
```python
|
||||
# Define user data
|
||||
USER_DATA = """
|
||||
About me
|
||||
- I'm Deshraj Yadav, Co-founder and CTO at Mem0, interested in AI and ML Infrastructure.
|
||||
- Previously, I was a Senior Autopilot Engineer at Tesla, leading the AI Platform for Autopilot.
|
||||
- I built EvalAI at Georgia Tech, an open-source platform for evaluating ML algorithms.
|
||||
- Outside of work, I enjoy playing cricket in two leagues in the San Francisco.
|
||||
"""
|
||||
|
||||
# Add user data to memory
|
||||
memory.add(USER_DATA, user_id=USER_ID)
|
||||
print("User data added to memory.")
|
||||
```
|
||||
|
||||
## Retrieving Relevant Memories
|
||||
|
||||
Now, we'll define our search command and retrieve relevant memories from Mem0.
|
||||
|
||||
```python
|
||||
# Define search command and retrieve relevant memories
|
||||
command = "Find papers on arxiv that I should read based on my interests."
|
||||
|
||||
relevant_memories = memory.search(command, user_id=USER_ID, limit=3)
|
||||
relevant_memories_text = '\n'.join(mem['text'] for mem in relevant_memories)
|
||||
print(f"Relevant memories:")
|
||||
print(relevant_memories_text)
|
||||
```
|
||||
|
||||
## Browsing arXiv
|
||||
|
||||
Finally, we'll use MultiOn to browse arXiv based on our command and relevant memories.
|
||||
|
||||
```python
|
||||
# Create prompt and browse arXiv
|
||||
prompt = f"{command}\n My past memories: {relevant_memories_text}"
|
||||
browse_result = multion.browse(cmd=prompt, url="https://arxiv.org/")
|
||||
print(browse_result)
|
||||
```
|
||||
|
||||
## Conclusion
|
||||
|
||||
By integrating Mem0 with MultiOn, you've created a personalized browser agent that remembers user preferences and automates web tasks. For more details and advanced usage, refer to the full [cookbook here](https://github.com/mem0ai/mem0/blob/main/cookbooks/mem0-multion.ipynb).
|
||||
@@ -1,56 +0,0 @@
|
||||
---
|
||||
title: 📚 Introduction
|
||||
description: '📝 Embedchain is a framework to easily create LLM powered bots over any dataset.'
|
||||
---
|
||||
|
||||
## 🤔 What is Embedchain?
|
||||
|
||||
Embedchain abstracts the entire process of loading a dataset, chunking it, creating embeddings, and storing it in a vector database.
|
||||
|
||||
You can add a single or multiple datasets using the `.add` method. Then, simply use the `.query` method to find answers from the added datasets.
|
||||
|
||||
If you want to create a Naval Ravikant bot with a YouTube video, a book in PDF format, two blog posts, and a question and answer pair, all you need to do is add the respective links. Embedchain will take care of the rest, creating a bot for you.
|
||||
|
||||
```python
|
||||
from embedchain import App
|
||||
|
||||
naval_chat_bot = App()
|
||||
# Embed Online Resources
|
||||
naval_chat_bot.add("https://www.youtube.com/watch?v=3qHkcs3kG44")
|
||||
naval_chat_bot.add("https://navalmanack.s3.amazonaws.com/Eric-Jorgenson_The-Almanack-of-Naval-Ravikant_Final.pdf")
|
||||
naval_chat_bot.add("https://nav.al/feedback")
|
||||
naval_chat_bot.add("https://nav.al/agi")
|
||||
|
||||
# Embed Local Resources
|
||||
naval_chat_bot.add(("Who is Naval Ravikant?", "Naval Ravikant is an Indian-American entrepreneur and investor."))
|
||||
|
||||
naval_chat_bot.query("What unique capacity does Naval argue humans possess when it comes to understanding explanations or concepts?")
|
||||
# Answer: Naval argues that humans possess the unique capacity to understand explanations or concepts to the maximum extent possible in this physical reality.
|
||||
```
|
||||
|
||||
## 🚀 How it works?
|
||||
|
||||
Creating a chat bot over any dataset involves the following steps:
|
||||
|
||||
1. Detect the data type and load the data
|
||||
2. Create meaningful chunks
|
||||
3. Create embeddings for each chunk
|
||||
4. Store the chunks in a vector database
|
||||
|
||||
When a user asks a query, the following process happens to find the answer:
|
||||
|
||||
1. Create an embedding for the query
|
||||
2. Find similar documents for the query from the vector database
|
||||
3. Pass the similar documents as context to LLM to get the final answer.
|
||||
|
||||
The process of loading the dataset and querying involves multiple steps, each with its own nuances:
|
||||
|
||||
- How should I chunk the data? What is a meaningful chunk size?
|
||||
- How should I create embeddings for each chunk? Which embedding model should I use?
|
||||
- How should I store the chunks in a vector database? Which vector database should I use?
|
||||
- Should I store metadata along with the embeddings?
|
||||
- How should I find similar documents for a query? Which ranking model should I use?
|
||||
|
||||
Embedchain takes care of all these nuances and provides a simple interface to create bots over any dataset.
|
||||
|
||||
In the first release, we make it easier for anyone to get a chatbot over any dataset up and running in less than a minute. Just create an app instance, add the datasets using the `.add` method, and use the `.query` method to get the relevant answers.
|
||||
@@ -0,0 +1,125 @@
|
||||
---
|
||||
title: 🤖 Overview
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Mem0 includes built-in support for various popular large language models. Memory can utilize the LLM provided by the user, ensuring efficient use for specific needs.
|
||||
|
||||
<CardGroup cols={4}>
|
||||
<Card title="OpenAI" href="#openai"></Card>
|
||||
<Card title="Groq" href="#groq"></Card>
|
||||
<Card title="Together" href="#together"></Card>
|
||||
<Card title="AWS Bedrock" href="#aws_bedrock"></Card>
|
||||
</CardGroup>
|
||||
|
||||
## OpenAI
|
||||
|
||||
To use OpenAI LLM models, you have to set the `OPENAI_API_KEY` environment variable. You can obtain the OpenAI API key from the [OpenAI Platform](https://platform.openai.com/account/api-keys).
|
||||
|
||||
Once you have obtained the key, you can use it like this:
|
||||
|
||||
```python
|
||||
import os
|
||||
from mem0 import Memory
|
||||
|
||||
os.environ['OPENAI_API_KEY'] = 'xxx'
|
||||
|
||||
config = {
|
||||
"llm": {
|
||||
"provider": "openai",
|
||||
"config": {
|
||||
"model": "gpt-4o",
|
||||
"temperature": 0.2,
|
||||
"max_tokens": 1500,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
m = Memory.from_config(config)
|
||||
m.add("Likes to play cricket on weekends", user_id="alice", metadata={"category": "hobbies"})
|
||||
```
|
||||
|
||||
## Groq
|
||||
|
||||
[Groq](https://groq.com/) is the creator of the world's first Language Processing Unit (LPU), providing exceptional speed performance for AI workloads running on their LPU Inference Engine.
|
||||
|
||||
In order to use LLMs from Groq, go to their [platform](https://console.groq.com/keys) and get the API key. Set the API key as `GROQ_API_KEY` environment variable to use the model as given below in the example.
|
||||
|
||||
```python
|
||||
import os
|
||||
from mem0 import Memory
|
||||
|
||||
os.environ['GROQ_API_KEY'] = 'xxx'
|
||||
|
||||
config = {
|
||||
"llm": {
|
||||
"provider": "groq",
|
||||
"config": {
|
||||
"model": "mixtral-8x7b-32768",
|
||||
"temperature": 0.1,
|
||||
"max_tokens": 1000,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
m = Memory.from_config(config)
|
||||
m.add("Likes to play cricket on weekends", user_id="alice", metadata={"category": "hobbies"})
|
||||
```
|
||||
|
||||
## TogetherAI
|
||||
|
||||
To use TogetherAI LLM models, you have to set the `TOGETHER_API_KEY` environment variable. You can obtain the TogetherAI API key from their [Account settings page](https://api.together.xyz/settings/api-keys).
|
||||
|
||||
Once you have obtained the key, you can use it like this:
|
||||
|
||||
```python
|
||||
import os
|
||||
from mem0 import Memory
|
||||
|
||||
os.environ['TOGETHER_API_KEY'] = 'xxx'
|
||||
|
||||
config = {
|
||||
"llm": {
|
||||
"provider": "togetherai",
|
||||
"config": {
|
||||
"model": "mistralai/Mixtral-8x7B-Instruct-v0.1",
|
||||
"temperature": 0.2,
|
||||
"max_tokens": 1500,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
m = Memory.from_config(config)
|
||||
m.add("Likes to play cricket on weekends", user_id="alice", metadata={"category": "hobbies"})
|
||||
```
|
||||
|
||||
## AWS Bedrock
|
||||
|
||||
### Setup
|
||||
- Before using the AWS Bedrock LLM, make sure you have the appropriate model access from [Bedrock Console](https://us-east-1.console.aws.amazon.com/bedrock/home?region=us-east-1#/modelaccess).
|
||||
- You will also need to authenticate the `boto3` client by using a method in the [AWS documentation](https://boto3.amazonaws.com/v1/documentation/api/latest/guide/credentials.html#configuring-credentials)
|
||||
- You will have to export `AWS_REGION`, `AWS_ACCESS_KEY`, and `AWS_SECRET_ACCESS_KEY` to set environment variables.
|
||||
|
||||
```python
|
||||
import os
|
||||
from mem0 import Memory
|
||||
|
||||
os.environ['AWS_REGION'] = 'us-east-1'
|
||||
os.environ["AWS_ACCESS_KEY"] = "xx"
|
||||
os.environ["AWS_SECRET_ACCESS_KEY"] = "xx"
|
||||
|
||||
config = {
|
||||
"llm": {
|
||||
"provider": "aws_bedrock",
|
||||
"config": {
|
||||
"model": "arn:aws:bedrock:us-east-1:123456789012:model/your-model-name",
|
||||
"temperature": 0.2,
|
||||
"max_tokens": 1500,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
m = Memory.from_config(config)
|
||||
m.add("Likes to play cricket on weekends", user_id="alice", metadata={"category": "hobbies"})
|
||||
```
|
||||
|
Before Width: | Height: | Size: 42 KiB After Width: | Height: | Size: 13 KiB |
|
After Width: | Height: | Size: 66 KiB |
|
Before Width: | Height: | Size: 42 KiB After Width: | Height: | Size: 13 KiB |
@@ -1,55 +1,94 @@
|
||||
{
|
||||
"$schema": "https://mintlify.com/schema.json",
|
||||
"name": "Embedchain",
|
||||
"name": "Mem0.ai",
|
||||
"favicon": "/logo/favicon.png",
|
||||
"colors": {
|
||||
"primary": "#3B2FC9",
|
||||
"light": "#6673FF",
|
||||
"dark": "#3B2FC9",
|
||||
"background": {
|
||||
"dark": "#0f1117",
|
||||
"light": "#fff"
|
||||
}
|
||||
},
|
||||
"logo": {
|
||||
"dark": "/logo/dark.svg",
|
||||
"light": "/logo/light.svg"
|
||||
},
|
||||
"favicon": "/favicon.png",
|
||||
"colors": {
|
||||
"primary": "#12A7D3",
|
||||
"light": "#81D7F7",
|
||||
"dark": "#004E7A"
|
||||
"light": "/logo/light.svg",
|
||||
"href": "https://github.com/embedchain/embedchain"
|
||||
},
|
||||
"tabs": [
|
||||
{
|
||||
"name": "💡 Examples",
|
||||
"url": "examples"
|
||||
},
|
||||
{
|
||||
"name": "🖥️ Platform",
|
||||
"url": "platform"
|
||||
}
|
||||
],
|
||||
"topbarLinks": [
|
||||
{
|
||||
"name": "Twitter",
|
||||
"url": "https://twitter.com/embedchain"
|
||||
"name": "Support",
|
||||
"url": "mailto:founders@mem0.ai"
|
||||
}
|
||||
],
|
||||
"anchors": [
|
||||
{
|
||||
"name": "Slack",
|
||||
"icon": "slack",
|
||||
"url": "https://mem0.ai/slack/"
|
||||
},
|
||||
{
|
||||
"name": "Discord",
|
||||
"url": "https://discord.gg/6PzXDgEjG5"
|
||||
"icon": "discord",
|
||||
"url": "https://mem0.ai/discord/"
|
||||
},
|
||||
{
|
||||
"name": "Talk to founders",
|
||||
"icon": "calendar",
|
||||
"url": "https://cal.com/taranjeetio/meet"
|
||||
}
|
||||
],
|
||||
"topbarCtaButton": {
|
||||
"name": "GitHub",
|
||||
"url": "https://embedchain.ai"
|
||||
},
|
||||
"navigation": [
|
||||
{
|
||||
"group": "Getting started",
|
||||
"pages": ["quickstart", "introduction"]
|
||||
"group": "Get Started",
|
||||
"pages": [
|
||||
"overview",
|
||||
"quickstart"
|
||||
]
|
||||
},
|
||||
{
|
||||
"group": "Advanced",
|
||||
"pages": ["advanced/app_types", "advanced/interface_types", "advanced/adding_data", "advanced/data_types", "advanced/query_configuration", "advanced/configuration", "advanced/testing", "advanced/vector_database", "advanced/showcase"]
|
||||
"group": "LLMs",
|
||||
"pages": [
|
||||
"llms"
|
||||
]
|
||||
},
|
||||
{
|
||||
"group": "Examples",
|
||||
"pages": ["examples/full_stack", "examples/api_server", "examples/discord_bot", "examples/slack_bot", "examples/telegram_bot", "examples/whatsapp_bot", "examples/poe_bot"]
|
||||
"group": "Integrations",
|
||||
"pages": [
|
||||
"integrations/multion"
|
||||
]
|
||||
},
|
||||
{
|
||||
"group": "Contribution Guidelines",
|
||||
"pages": ["contribution/dev", "contribution/docs"]
|
||||
"group": "💡 Examples",
|
||||
"pages": [
|
||||
"examples/overview",
|
||||
"examples/personal-ai-tutor",
|
||||
"examples/customer-support-agent",
|
||||
"examples/personal-travel-assistant"
|
||||
]
|
||||
},
|
||||
{
|
||||
"group": "🖥️ Platform",
|
||||
"pages": [
|
||||
"platform/overview",
|
||||
"platform/quickstart"
|
||||
]
|
||||
}
|
||||
|
||||
],
|
||||
"footerSocials": {
|
||||
"twitter": "https://twitter.com/embedchain",
|
||||
"github": "https://github.com/embedchain/embedchain",
|
||||
"linkedin": "https://www.linkedin.com/company/embedchain",
|
||||
"website": "https://embedchain.ai"
|
||||
},
|
||||
"backgroundImage": "/background.png",
|
||||
"isWhiteLabeled": true
|
||||
}
|
||||
"x": "https://x.com/mem0ai",
|
||||
"github": "https://github.com/embedchain/embedchain/mem0",
|
||||
"linkedin": "https://www.linkedin.com/company/mem0/"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
---
|
||||
title: 📚 Overview
|
||||
description: 'Welcome to the Mem0 docs!'
|
||||
---
|
||||
|
||||
> Mem0 provides a smart, self-improving memory layer for Large Language Models, enabling personalized AI experiences across applications.
|
||||
|
||||
## Core features
|
||||
|
||||
- **User, Session, and AI Agent Memory**: Retains information across user sessions, interactions, and AI agents, ensuring continuity and context.
|
||||
- **Adaptive Personalization**: Continuously improves personalization based on user interactions and feedback.
|
||||
- **Developer-Friendly API**: Offers a straightforward API for seamless integration into various applications.
|
||||
- **Platform Consistency**: Ensures consistent behavior and data across different platforms and devices.
|
||||
- **Managed Service**: Provides a hosted solution for easy deployment and maintenance.
|
||||
|
||||
If you are looking to quick start, jump to one of the following links:
|
||||
|
||||
<CardGroup cols={2}>
|
||||
<Card title="Quickstart" icon="square-1" href="/quickstart/">
|
||||
Jump to quickstart section to get started
|
||||
</Card>
|
||||
<Card title="Examples" icon="square-2" href="/examples/overview/">
|
||||
Checkout curated examples
|
||||
</Card>
|
||||
</CardGroup>
|
||||
|
||||
## Common Use Cases
|
||||
|
||||
- **Personalized Learning Assistants**: Long-term memory allows learning assistants to remember user preferences, past interactions, and progress, providing a more tailored and effective learning experience.
|
||||
|
||||
- **Customer Support AI Agents**: By retaining information from previous interactions, customer support bots can offer more accurate and context-aware assistance, improving customer satisfaction and reducing resolution times.
|
||||
|
||||
- **Healthcare Assistants**: Long-term memory enables healthcare assistants to keep track of patient history, medication schedules, and treatment plans, ensuring personalized and consistent care.
|
||||
|
||||
- **Virtual Companions**: Virtual companions can use long-term memory to build deeper relationships with users by remembering personal details, preferences, and past conversations, making interactions more meaningful.
|
||||
|
||||
- **Productivity Tools**: Long-term memory helps productivity tools remember user habits, frequently used documents, and task history, streamlining workflows and enhancing efficiency.
|
||||
|
||||
- **Gaming AI**: In gaming, AI with long-term memory can create more immersive experiences by remembering player choices, strategies, and progress, adapting the game environment accordingly.
|
||||
|
||||
## How is Mem0 different from RAG?
|
||||
|
||||
Mem0's memory implementation for Large Language Models (LLMs) offers several advantages over Retrieval-Augmented Generation (RAG):
|
||||
|
||||
- **Entity Relationships**: Mem0 can understand and relate entities across different interactions, unlike RAG which retrieves information from static documents. This leads to a deeper understanding of context and relationships.
|
||||
|
||||
- **Recency, Relevancy, and Decay**: Mem0 prioritizes recent interactions and gradually forgets outdated information, ensuring the memory remains relevant and up-to-date for more accurate responses.
|
||||
|
||||
- **Contextual Continuity**: Mem0 retains information across sessions, maintaining continuity in conversations and interactions, which is essential for long-term engagement applications like virtual companions or personalized learning assistants.
|
||||
|
||||
- **Adaptive Learning**: Mem0 improves its personalization based on user interactions and feedback, making the memory more accurate and tailored to individual users over time.
|
||||
|
||||
- **Dynamic Updates**: Mem0 can dynamically update its memory with new information and interactions, unlike RAG which relies on static data. This allows for real-time adjustments and improvements, enhancing the user experience.
|
||||
|
||||
These advanced memory capabilities make Mem0 a powerful tool for developers aiming to create personalized and context-aware AI applications.
|
||||
|
||||
If you have any questions, please feel free to reach out to us using one of the following methods:
|
||||
|
||||
<Snippet file="get-help.mdx" />
|
||||
@@ -0,0 +1,45 @@
|
||||
---
|
||||
title: Introduction
|
||||
description: 'Empower your AI applications with long-term memory and personalization'
|
||||
---
|
||||
|
||||
## Welcome to Mem0 Platform
|
||||
|
||||
Mem0 Platform is a managed service that revolutionizes the way AI applications handle memory. By providing a smart, self-improving memory layer for Large Language Models (LLMs), we enable developers to create personalized AI experiences that evolve with each user interaction.
|
||||
|
||||
## Why Choose Mem0 Platform?
|
||||
|
||||
1. **Enhanced User Experience**: Deliver tailored interactions that make your AI applications truly stand out.
|
||||
2. **Simplified Development**: Our API-first approach streamlines integration, allowing you to focus on building great features.
|
||||
3. **Scalable Solution**: Designed to grow with your application, from prototypes to production-ready systems.
|
||||
|
||||
## Key Features
|
||||
|
||||
- **Comprehensive Memory Management**: Easily manage long-term, short-term, semantic, and episodic memories for individual users, agents, and sessions through our robust APIs.
|
||||
- **Self-Improving Memory**: Our adaptive system continuously learns from user interactions, refining its understanding over time.
|
||||
- **Cross-Platform Consistency**: Ensure a unified user experience across various AI platforms and applications.
|
||||
- **Centralized Memory Control**: Store, update, and delete memories effortlessly, taking away the hassle of memory management.
|
||||
|
||||
## Common Use Cases
|
||||
|
||||
- Personalized Learning Assistants
|
||||
- Customer Support AI Agents
|
||||
- Healthcare Assistants
|
||||
- Virtual Companions
|
||||
- Productivity Tools
|
||||
- Gaming AI
|
||||
|
||||
## Getting Started
|
||||
Ready to supercharge your AI application with Mem0? Follow these steps:
|
||||
|
||||
1. **Sign Up**: Create your Mem0 account at our platform.
|
||||
2. **API Key**: Generate your API key in the dashboard.
|
||||
3. **Installation**: Install our Python SDK using pip: `pip install mem0ai`
|
||||
4. **Quick Implementation**: Check out our [Quickstart Guide](/platform/quickstart) to start using Mem0 quickly.
|
||||
|
||||
## Next Steps
|
||||
|
||||
- Explore our API Reference for detailed endpoint documentation.
|
||||
- Join our [slack](https://mem0.ai/slack) or [discord](https://mem0.ai/discord) with other developers and get support.
|
||||
|
||||
We're excited to see what you'll build with Mem0 Platform. Let's create smarter, more personalized AI experiences together!
|
||||
@@ -0,0 +1,358 @@
|
||||
---
|
||||
title: Quickstart
|
||||
description: 'Get started with Mem0 Platform in minutes'
|
||||
---
|
||||
|
||||
## 1. Installation
|
||||
|
||||
Install the Mem0 Python package:
|
||||
|
||||
```bash
|
||||
pip install mem0ai
|
||||
```
|
||||
|
||||
## 2. API Key Setup
|
||||
|
||||
1. Sign in to [Mem0 Platform](https://app.mem0.ai/dashboard/api-keys)
|
||||
2. Copy your API Key from the dashboard
|
||||
|
||||

|
||||
|
||||
## 3. Instantiate Client
|
||||
|
||||
```python
|
||||
from mem0 import MemoryClient
|
||||
client = MemoryClient(api_key="your-api-key")
|
||||
```
|
||||
|
||||
## 4. Memory Operations
|
||||
|
||||
We provide a simple yet customizable interface for performing CRUD operations on memory. Here is how you can create and get memories:
|
||||
|
||||
|
||||
### 4.1 Create Memories
|
||||
|
||||
For users (long-term memory):
|
||||
|
||||
```python
|
||||
# create long-term memory for users
|
||||
client.add("Remember my name is Deshraj Yadav.", user_id="deshraj")
|
||||
client.add("I like to eat pizza and go out on weekends.", user_id="deshraj")
|
||||
client.add("Oh I am actually allergic to cheese to cannot eat pizza anymore.", user_id="deshraj")
|
||||
```
|
||||
|
||||
Output:
|
||||
```python
|
||||
{'message': 'Memory added successfully!'}
|
||||
```
|
||||
|
||||
You can see all the memory operations happening on the platform itself.
|
||||
|
||||

|
||||
|
||||
|
||||
You can also add memories for a particular session or for an AI agent that you are building:
|
||||
|
||||
- For user sessions (short-term memory):
|
||||
|
||||
```python
|
||||
client.add("Deshraj is building Gmail AI agent", user_id="deshraj", session_id="session-1")
|
||||
```
|
||||
|
||||
- For agents (long-term memory):
|
||||
|
||||
```python
|
||||
client.add("Return short responses when responding to emails", agent_id="gmail-agent")
|
||||
```
|
||||
|
||||
### 4.2 Retrieve Memories
|
||||
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python Code
|
||||
client.get_all(user_id="deshraj")
|
||||
```
|
||||
|
||||
```python Output
|
||||
[
|
||||
{
|
||||
'id': 'dbce6e06-6adf-40b8-9187-3d30bd13b741',
|
||||
'agent': None,
|
||||
'consumer': {
|
||||
'id': 8,
|
||||
'user_id': 'deshraj',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:23.899900-07:00',
|
||||
'updated_at': '2024-07-17T16:47:23.899918-07:00'
|
||||
},
|
||||
'app': None,
|
||||
'run': None,
|
||||
'hash': '57288ac8a87c4ac8d3ac7f2075d264ca',
|
||||
'input': 'Remember my name is Deshraj Yadav.',
|
||||
'text': 'My name is Deshraj Yadav.',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:25.670180-07:00',
|
||||
'updated_at': '2024-07-17T16:47:25.670197-07:00'
|
||||
},
|
||||
{
|
||||
'id': 'f6dec5d1-b5db-45f5-a2fb-3979a0f27d30',
|
||||
'agent': None,
|
||||
'consumer': {
|
||||
'id': 8,
|
||||
'user_id': 'deshraj',
|
||||
# ... other consumer fields ...
|
||||
},
|
||||
# ... other fields ...
|
||||
'text': 'I am allergic to cheese so I cannot eat pizza anymore.',
|
||||
# ... remaining fields ...
|
||||
},
|
||||
# ... additional memory entries ...
|
||||
]
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
Similarly, you can get all memories for an agent:
|
||||
|
||||
```python
|
||||
agent_memories = client.get_all(agent_id="gmail-agent")
|
||||
```
|
||||
|
||||
Get specific memory:
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python Code
|
||||
memory = client.get(memory_id="dbce6e06-6adf-40b8-9187-3d30bd13b741")
|
||||
```
|
||||
|
||||
```python Output
|
||||
{
|
||||
'id': 'dbce6e06-6adf-40b8-9187-3d30bd13b741',
|
||||
'agent': None,
|
||||
'consumer': {
|
||||
'id': 8,
|
||||
'user_id': 'deshraj',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:23.899900-07:00',
|
||||
'updated_at': '2024-07-17T16:47:23.899918-07:00'
|
||||
},
|
||||
'app': None,
|
||||
'run': None,
|
||||
'hash': '57288ac8a87c4ac8d3ac7f2075d264ca',
|
||||
'input': 'Remember my name is Deshraj Yadav.',
|
||||
'text': 'My name is Deshraj Yadav.',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:25.670180-07:00',
|
||||
'updated_at': '2024-07-17T16:47:25.670197-07:00'
|
||||
}
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
### 4.3 Update Memory
|
||||
|
||||
You can also update specific memory by using the following method:
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python Code
|
||||
client.update(memory_id, data="Updated name is Deshraj Kumar")
|
||||
```
|
||||
|
||||
```python Output
|
||||
{
|
||||
'id': 'dbce6e06-6adf-40b8-9187-3d30bd13b741',
|
||||
'agent': None,
|
||||
'consumer': {
|
||||
'id': 8,
|
||||
'user_id': 'deshraj',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:23.899900-07:00',
|
||||
'updated_at': '2024-07-17T16:47:23.899918-07:00'
|
||||
},
|
||||
'app': None,
|
||||
'run': None,
|
||||
'hash': '57288ac8a87c4ac8d3ac7f2075d264ca',
|
||||
'input': 'Updated name is Deshraj Kumar.',
|
||||
'text': 'Name is Deshraj Kumar.',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:25.670180-07:00',
|
||||
'updated_at': '2024-07-17T16:47:25.670197-07:00'
|
||||
}
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
|
||||
### 4.4 Memory History
|
||||
|
||||
Get history of how a memory has changed over time
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python Code
|
||||
history = client.history(memory_id)
|
||||
```
|
||||
|
||||
```python Output
|
||||
[
|
||||
{
|
||||
'id': '51193804-2ee6-4f81-b4e7-497e98b70858',
|
||||
'memory': {
|
||||
'id': 'dbce6e06-6adf-40b8-9187-3d30bd13b741',
|
||||
'agent': None,
|
||||
'consumer': {
|
||||
'id': 8,
|
||||
'user_id': 'deshraj',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:23.899900-07:00',
|
||||
'updated_at': '2024-07-17T16:47:23.899918-07:00'
|
||||
},
|
||||
'app': None,
|
||||
'run': None,
|
||||
'hash': '57288ac8a87c4ac8d3ac7f2075d264ca',
|
||||
'input': 'Remember my name is Deshraj Yadav.',
|
||||
'text': 'My name is Deshraj Yadav.',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:25.670180-07:00',
|
||||
'updated_at': '2024-07-17T16:47:25.670197-07:00'
|
||||
},
|
||||
'hash': '57288ac8a87c4ac8d3ac7f2075d264ca',
|
||||
'event': 'ADD',
|
||||
'input': 'Remember my name is Deshraj Yadav.',
|
||||
'previous_text': None,
|
||||
'text': 'My name is Deshraj Yadav.',
|
||||
'metadata': None,
|
||||
'created_at': '2024-07-17T16:47:25.686899-07:00',
|
||||
'updated_at': '2024-07-17T16:47:25.670197-07:00',
|
||||
'change_description': 'Memory ADD event'
|
||||
}
|
||||
]
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
### 4.5 Search for relevant memories
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python Code
|
||||
client.search("What does Deshraj like to eat?", user_id="deshraj", limit=3)
|
||||
```
|
||||
|
||||
```python Output
|
||||
[
|
||||
{
|
||||
"id": "dbce6e06-6adf-40b8-9187-3d30bd13b741",
|
||||
"agent": null,
|
||||
"consumer": {
|
||||
"id": 8,
|
||||
"user_id": "deshraj",
|
||||
"metadata": null,
|
||||
"created_at": "...",
|
||||
"updated_at": "..."
|
||||
},
|
||||
"app": null,
|
||||
"run": null,
|
||||
"hash": "57288ac8a87c4ac8d3ac7f2075d264ca",
|
||||
"input": "Remember my name is Deshraj Yadav.",
|
||||
"text": "My name is Deshraj Yadav.",
|
||||
"metadata": null,
|
||||
"created_at": "2024-07-17T16:47:25.670180-07:00",
|
||||
"updated_at": "..."
|
||||
},
|
||||
{
|
||||
"id": "091dbed6-74b4-4e15-b765-81be2abe0d6b",
|
||||
"agent": null,
|
||||
"consumer": {
|
||||
"id": 8,
|
||||
"user_id": "deshraj",
|
||||
"metadata": null,
|
||||
"created_at": "...",
|
||||
"updated_at": "..."
|
||||
},
|
||||
"app": null,
|
||||
"run": null,
|
||||
"hash": "622a5a24d5ac54136414a22ec12f9520",
|
||||
"input": "Oh I am actually allergic to cheese to cannot eat pizza anymore.",
|
||||
"text": "I like to eat pizza and go out on weekends.",
|
||||
"metadata": null,
|
||||
"created_at": "2024-07-17T16:49:24.276695-07:00",
|
||||
"updated_at": "..."
|
||||
},
|
||||
{
|
||||
"id": "5fb8f85d-3383-4bad-9d46-f171272478a4",
|
||||
"agent": null,
|
||||
"consumer": {
|
||||
"id": 8,
|
||||
"user_id": "deshraj",
|
||||
"metadata": null,
|
||||
"created_at": "...",
|
||||
"updated_at": "..."
|
||||
},
|
||||
"app": null,
|
||||
"run": {
|
||||
"id": 1,
|
||||
"run_id": "session-1",
|
||||
"name": "",
|
||||
"metadata": null,
|
||||
"created_at": "...",
|
||||
"updated_at": "..."
|
||||
},
|
||||
"hash": "179ced9649ac2b85350ece4946b1ee9b",
|
||||
"input": "Deshraj is building Gmail AI agent",
|
||||
"text": "Deshraj is building Gmail AI agent",
|
||||
"metadata": null,
|
||||
"created_at": "2024-07-17T16:52:41.278920-07:00",
|
||||
"updated_at": "..."
|
||||
},
|
||||
{
|
||||
"id": "f6dec5d1-b5db-45f5-a2fb-3979a0f27d30",
|
||||
"agent": null,
|
||||
"consumer": {
|
||||
"id": 8,
|
||||
"user_id": "deshraj",
|
||||
"metadata": null,
|
||||
"created_at": "...",
|
||||
"updated_at": "..."
|
||||
},
|
||||
"app": null,
|
||||
"run": null,
|
||||
"hash": "19248f0766044b5973fc0ef1bf3955ef",
|
||||
"input": "Oh I am actually allergic to cheese to cannot eat pizza anymore.",
|
||||
"text": "I am allergic to cheese so I cannot eat pizza anymore.",
|
||||
"metadata": null,
|
||||
"created_at": "2024-07-17T16:49:38.622084-07:00",
|
||||
"updated_at": "..."
|
||||
}
|
||||
]
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
|
||||
### 4.6 Delete Memory
|
||||
|
||||
Delete specific memory:
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python Code
|
||||
client.delete(memory_id)
|
||||
```
|
||||
|
||||
```python Output
|
||||
{'message': 'Memory deleted successfully!'}
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
|
||||
Delete all memories of a user:
|
||||
|
||||
<CodeGroup>
|
||||
|
||||
```python Code
|
||||
client.delete_all(user_id="alex")
|
||||
```
|
||||
|
||||
```python Output
|
||||
{'message': 'Memories deleted successfully!'}
|
||||
```
|
||||
</CodeGroup>
|
||||
@@ -1,35 +1,210 @@
|
||||
---
|
||||
title: '🚀 Quickstart'
|
||||
description: '💡 Start building LLM powered bots under 30 seconds'
|
||||
title: 🚀 Quickstart
|
||||
description: 'Get started with Mem0 quickly!'
|
||||
---
|
||||
|
||||
Install embedchain python package:
|
||||
> Welcome to the Mem0 quickstart guide. This guide will help you get up and running with Mem0 in no time.
|
||||
|
||||
## Installation
|
||||
|
||||
To install Mem0, you can use pip. Run the following command in your terminal:
|
||||
|
||||
```bash
|
||||
pip install embedchain
|
||||
pip install mem0ai
|
||||
```
|
||||
|
||||
Creating a chatbot involves 3 steps:
|
||||
## Basic Usage
|
||||
|
||||
- ⚙️ Import the App instance
|
||||
- 🗃️ Add Dataset
|
||||
- 💬 Query or Chat on the dataset and get answers (Interface Types)
|
||||
### Initialize Mem0
|
||||
|
||||
Run your first bot in python using the following code. Make sure to set the `OPENAI_API_KEY` 🔑 environment variable in the code.
|
||||
<Tabs>
|
||||
<Tab title="Basic">
|
||||
```python
|
||||
from mem0 import Memory
|
||||
m = Memory()
|
||||
```
|
||||
</Tab>
|
||||
<Tab title="Advanced">
|
||||
If you want to run Mem0 in production, initialize using the following method:
|
||||
|
||||
Run Qdrant first:
|
||||
|
||||
```bash
|
||||
docker pull qdrant/qdrant
|
||||
|
||||
docker run -p 6333:6333 -p 6334:6334 \
|
||||
-v $(pwd)/qdrant_storage:/qdrant/storage:z \
|
||||
qdrant/qdrant
|
||||
```
|
||||
|
||||
Then, instantiate memory with qdrant server:
|
||||
|
||||
```python
|
||||
import os
|
||||
from mem0 import Memory
|
||||
|
||||
from embedchain import App
|
||||
config = {
|
||||
"vector_store": {
|
||||
"provider": "qdrant",
|
||||
"config": {
|
||||
"host": "localhost",
|
||||
"port": 6333,
|
||||
}
|
||||
},
|
||||
}
|
||||
|
||||
os.environ["OPENAI_API_KEY"] = "xxx"
|
||||
elon_musk_bot = App()
|
||||
|
||||
# Embed Online Resources
|
||||
elon_musk_bot.add("https://en.wikipedia.org/wiki/Elon_Musk")
|
||||
elon_musk_bot.add("https://www.tesla.com/elon-musk")
|
||||
|
||||
response = elon_musk_bot.query("How many companies does Elon Musk run?")
|
||||
print(response)
|
||||
# Answer: 'Elon Musk runs four companies: Tesla, SpaceX, Neuralink, and The Boring Company.'
|
||||
m = Memory.from_config(config)
|
||||
```
|
||||
</Tab>
|
||||
</Tabs>
|
||||
|
||||
|
||||
### Store a Memory
|
||||
|
||||
```python
|
||||
# For a user
|
||||
result = m.add("Likes to play cricket on weekends", user_id="alice", metadata={"category": "hobbies"})
|
||||
print(result)
|
||||
```
|
||||
|
||||
Output:
|
||||
```python
|
||||
[
|
||||
{
|
||||
'id': 'm1',
|
||||
'event': 'add',
|
||||
'data': 'Likes to play cricket on weekends'
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
### Retrieve Memories
|
||||
|
||||
```python
|
||||
# Get all memories
|
||||
all_memories = m.get_all()
|
||||
print(all_memories)
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```python
|
||||
[
|
||||
{
|
||||
'id': 'm1',
|
||||
'text': 'Likes to play cricket on weekends',
|
||||
'metadata': {
|
||||
'data': 'Likes to play cricket on weekends',
|
||||
'category': 'hobbies'
|
||||
}
|
||||
},
|
||||
# ... other memories ...
|
||||
]
|
||||
```
|
||||
|
||||
```python
|
||||
# Get a single memory by ID
|
||||
specific_memory = m.get("m1")
|
||||
print(specific_memory)
|
||||
```
|
||||
|
||||
Output:
|
||||
```python
|
||||
{
|
||||
'id': 'm1',
|
||||
'text': 'Likes to play cricket on weekends',
|
||||
'metadata': {
|
||||
'data': 'Likes to play cricket on weekends',
|
||||
'category': 'hobbies'
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Search Memories
|
||||
|
||||
```python
|
||||
related_memories = m.search(query="What are Alice's hobbies?", user_id="alice")
|
||||
print(related_memories)
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```python
|
||||
[
|
||||
{
|
||||
'id': 'm1',
|
||||
'text': 'Likes to play cricket on weekends',
|
||||
'metadata': {
|
||||
'data': 'Likes to play cricket on weekends',
|
||||
'category': 'hobbies'
|
||||
},
|
||||
'score': 0.85 # Similarity score
|
||||
},
|
||||
# ... other related memories ...
|
||||
]
|
||||
```
|
||||
|
||||
### Update a Memory
|
||||
|
||||
```python
|
||||
result = m.update(memory_id="m1", data="Likes to play tennis on weekends")
|
||||
print(result)
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```python
|
||||
{
|
||||
'id': 'm1',
|
||||
'event': 'update',
|
||||
'data': 'Likes to play tennis on weekends'
|
||||
}
|
||||
```
|
||||
|
||||
### Memory History
|
||||
|
||||
```python
|
||||
history = m.history(memory_id="m1")
|
||||
print(history)
|
||||
```
|
||||
Output:
|
||||
```python
|
||||
[
|
||||
{
|
||||
'id': 'h1',
|
||||
'memory_id': 'm1',
|
||||
'prev_value': None,
|
||||
'new_value': 'Likes to play cricket on weekends',
|
||||
'event': 'add',
|
||||
'timestamp': '2024-07-14 10:00:54.466687',
|
||||
'is_deleted': 0
|
||||
},
|
||||
{
|
||||
'id': 'h2',
|
||||
'memory_id': 'm1',
|
||||
'prev_value': 'Likes to play cricket on weekends',
|
||||
'new_value': 'Likes to play tennis on weekends',
|
||||
'event': 'update',
|
||||
'timestamp': '2024-07-14 10:15:17.230943',
|
||||
'is_deleted': 0
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
### Delete Memory
|
||||
|
||||
```python
|
||||
m.delete(memory_id="m1") # Delete a memory
|
||||
|
||||
m.delete_all(user_id="alice") # Delete all memories
|
||||
```
|
||||
|
||||
### Reset Memory
|
||||
|
||||
```python
|
||||
m.reset() # Reset all memories
|
||||
```
|
||||
|
||||
|
||||
If you have any questions, please feel free to reach out to us using one of the following methods:
|
||||
|
||||
<Snippet file="get-help.mdx" />
|
||||
@@ -0,0 +1,4 @@
|
||||
One of the core principles of software development is DRY (Don't Repeat
|
||||
Yourself). This is a principle that apply to documentation as
|
||||
well. If you find yourself repeating the same content in multiple places, you
|
||||
should consider creating a custom snippet to keep your content in sync.
|
||||
@@ -1,10 +1,10 @@
|
||||
# Contributing to embedchain
|
||||
|
||||
Let us make contributing easy, collaborative and fun.
|
||||
Let us make contribution easy, collaborative and fun.
|
||||
|
||||
## Submit your Contribution through PR
|
||||
|
||||
To make a contribution, follow the following steps:
|
||||
To make a contribution, follow these steps:
|
||||
|
||||
1. Fork and clone this repository
|
||||
2. Do the changes on your fork with dedicated feature branch `feature/f1`
|
||||
@@ -35,7 +35,7 @@ poetry shell
|
||||
|
||||
### 📌 Pre-commit
|
||||
|
||||
To ensure our standards, make sure to install pre-commit before star to contribute.
|
||||
To ensure our standards, make sure to install pre-commit before starting to contribute.
|
||||
|
||||
```bash
|
||||
pre-commit install
|
||||
@@ -51,7 +51,7 @@ make lint
|
||||
|
||||
Make sure that the linter does not report any errors or warnings before submitting a pull request.
|
||||
|
||||
### Code Format with `black`
|
||||
### Code Formatting with `black`
|
||||
|
||||
We use `black` to reformat the code by running the following command:
|
||||
|
||||
@@ -67,6 +67,10 @@ We use `pytest` to test our code. You can run the tests by running the following
|
||||
poetry run pytest
|
||||
```
|
||||
|
||||
|
||||
Several packages have been removed from Poetry to make the package lighter. Therefore, it is recommended to run `make install_all` to install the remaining packages and ensure all tests pass.
|
||||
|
||||
|
||||
Make sure that all tests pass before submitting a pull request.
|
||||
|
||||
## 🚀 Release Process
|
||||
@@ -0,0 +1,56 @@
|
||||
# Variables
|
||||
PYTHON := python3
|
||||
PIP := $(PYTHON) -m pip
|
||||
PROJECT_NAME := embedchain
|
||||
|
||||
# Targets
|
||||
.PHONY: install format lint clean test ci_lint ci_test coverage
|
||||
|
||||
install:
|
||||
poetry install
|
||||
|
||||
# TODO: use a more efficient way to install these packages
|
||||
install_all:
|
||||
poetry install --all-extras
|
||||
poetry run pip install pinecone-text pinecone-client langchain-anthropic "unstructured[local-inference, all-docs]" ollama langchain_together==0.1.3 \
|
||||
langchain_cohere==0.1.5 deepgram-sdk==3.2.7 langchain-huggingface psutil clarifai==10.0.1 flask==2.3.3 twilio==8.5.0 fastapi-poe==0.0.16 discord==2.3.2 \
|
||||
slack-sdk==3.21.3 huggingface_hub==0.23.0 gitpython==3.1.38 yt_dlp==2023.11.14 PyGithub==1.59.1 feedparser==6.0.10 newspaper3k==0.2.8 listparser==0.19 \
|
||||
modal==0.56.4329 dropbox==11.36.2 boto3==1.34.20 youtube-transcript-api==0.6.1 pytube==15.0.0 beautifulsoup4==4.12.3
|
||||
|
||||
install_es:
|
||||
poetry install --extras elasticsearch
|
||||
|
||||
install_opensearch:
|
||||
poetry install --extras opensearch
|
||||
|
||||
install_milvus:
|
||||
poetry install --extras milvus
|
||||
|
||||
shell:
|
||||
poetry shell
|
||||
|
||||
py_shell:
|
||||
poetry run python
|
||||
|
||||
format:
|
||||
$(PYTHON) -m black .
|
||||
$(PYTHON) -m isort .
|
||||
|
||||
clean:
|
||||
rm -rf dist build *.egg-info
|
||||
|
||||
lint:
|
||||
poetry run ruff .
|
||||
|
||||
build:
|
||||
poetry build
|
||||
|
||||
publish:
|
||||
poetry publish
|
||||
|
||||
# for example: make test file=tests/test_factory.py
|
||||
test:
|
||||
poetry run pytest $(file)
|
||||
|
||||
coverage:
|
||||
poetry run pytest --cov=$(PROJECT_NAME) --cov-report=xml
|
||||
@@ -0,0 +1,125 @@
|
||||
<p align="center">
|
||||
<img src="docs/logo/dark.svg" width="400px" alt="Embedchain Logo">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://pypi.org/project/embedchain/">
|
||||
<img src="https://img.shields.io/pypi/v/embedchain" alt="PyPI">
|
||||
</a>
|
||||
<a href="https://pepy.tech/project/embedchain">
|
||||
<img src="https://static.pepy.tech/badge/embedchain" alt="Downloads">
|
||||
</a>
|
||||
<a href="https://embedchain.ai/slack">
|
||||
<img src="https://img.shields.io/badge/slack-embedchain-brightgreen.svg?logo=slack" alt="Slack">
|
||||
</a>
|
||||
<a href="https://embedchain.ai/discord">
|
||||
<img src="https://dcbadge.vercel.app/api/server/6PzXDgEjG5?style=flat" alt="Discord">
|
||||
</a>
|
||||
<a href="https://twitter.com/embedchain">
|
||||
<img src="https://img.shields.io/twitter/follow/embedchain" alt="Twitter">
|
||||
</a>
|
||||
<a href="https://colab.research.google.com/drive/138lMWhENGeEu7Q1-6lNbNTHGLZXBBz_B?usp=sharing">
|
||||
<img src="https://colab.research.google.com/assets/colab-badge.svg" alt="Open in Colab">
|
||||
</a>
|
||||
<a href="https://codecov.io/gh/embedchain/embedchain">
|
||||
<img src="https://codecov.io/gh/embedchain/embedchain/graph/badge.svg?token=EMRRHZXW1Q" alt="codecov">
|
||||
</a>
|
||||
</p>
|
||||
|
||||
<hr />
|
||||
|
||||
## What is Embedchain?
|
||||
|
||||
Embedchain is an Open Source Framework for personalizing LLM responses. It makes it easy to create and deploy personalized AI apps. At its core, Embedchain follows the design principle of being *"Conventional but Configurable"* to serve both software engineers and machine learning engineers.
|
||||
|
||||
Embedchain streamlines the creation of personalized LLM applications, offering a seamless process for managing various types of unstructured data. It efficiently segments data into manageable chunks, generates relevant embeddings, and stores them in a vector database for optimized retrieval. With a suite of diverse APIs, it enables users to extract contextual information, find precise answers, or engage in interactive chat conversations, all tailored to their own data.
|
||||
|
||||
## 🔧 Quick install
|
||||
|
||||
### Python API
|
||||
|
||||
```bash
|
||||
pip install embedchain
|
||||
```
|
||||
|
||||
## ✨ Live demo
|
||||
|
||||
Checkout the [Chat with PDF](https://embedchain.ai/demo/chat-pdf) live demo we created using Embedchain. You can find the source code [here](https://github.com/embedchain/embedchain/tree/main/examples/chat-pdf).
|
||||
|
||||
## 🔍 Usage
|
||||
|
||||
<!-- Demo GIF or Image -->
|
||||
<p align="center">
|
||||
<img src="docs/images/cover.gif" width="900px" alt="Embedchain Demo">
|
||||
</p>
|
||||
|
||||
For example, you can create an Elon Musk bot using the following code:
|
||||
|
||||
```python
|
||||
import os
|
||||
from embedchain import App
|
||||
|
||||
# Create a bot instance
|
||||
os.environ["OPENAI_API_KEY"] = "<YOUR_API_KEY>"
|
||||
app = App()
|
||||
|
||||
# Embed online resources
|
||||
app.add("https://en.wikipedia.org/wiki/Elon_Musk")
|
||||
app.add("https://www.forbes.com/profile/elon-musk")
|
||||
|
||||
# Query the app
|
||||
app.query("How many companies does Elon Musk run and name those?")
|
||||
# Answer: Elon Musk currently runs several companies. As of my knowledge, he is the CEO and lead designer of SpaceX, the CEO and product architect of Tesla, Inc., the CEO and founder of Neuralink, and the CEO and founder of The Boring Company. However, please note that this information may change over time, so it's always good to verify the latest updates.
|
||||
```
|
||||
|
||||
You can also try it in your browser with Google Colab:
|
||||
|
||||
[](https://colab.research.google.com/drive/17ON1LPonnXAtLaZEebnOktstB_1cJJmh?usp=sharing)
|
||||
|
||||
## 📖 Documentation
|
||||
Comprehensive guides and API documentation are available to help you get the most out of Embedchain:
|
||||
|
||||
- [Introduction](https://docs.embedchain.ai/get-started/introduction#what-is-embedchain)
|
||||
- [Getting Started](https://docs.embedchain.ai/get-started/quickstart)
|
||||
- [Examples](https://docs.embedchain.ai/examples)
|
||||
- [Supported data types](https://docs.embedchain.ai/components/data-sources/overview)
|
||||
|
||||
## 🔗 Join the Community
|
||||
|
||||
* Connect with fellow developers by joining our [Slack Community](https://embedchain.ai/slack) or [Discord Community](https://embedchain.ai/discord).
|
||||
|
||||
* Dive into [GitHub Discussions](https://github.com/embedchain/embedchain/discussions), ask questions, or share your experiences.
|
||||
|
||||
## 🤝 Schedule a 1-on-1 Session
|
||||
|
||||
Book a [1-on-1 Session](https://cal.com/taranjeetio/ec) with the founders, to discuss any issues, provide feedback, or explore how we can improve Embedchain for you.
|
||||
|
||||
## 🌐 Contributing
|
||||
|
||||
Contributions are welcome! Please check out the issues on the repository, and feel free to open a pull request.
|
||||
For more information, please see the [contributing guidelines](CONTRIBUTING.md).
|
||||
|
||||
For more reference, please go through [Development Guide](https://docs.embedchain.ai/contribution/dev) and [Documentation Guide](https://docs.embedchain.ai/contribution/docs).
|
||||
|
||||
<a href="https://github.com/embedchain/embedchain/graphs/contributors">
|
||||
<img src="https://contrib.rocks/image?repo=embedchain/embedchain" />
|
||||
</a>
|
||||
|
||||
## Anonymous Telemetry
|
||||
|
||||
We collect anonymous usage metrics to enhance our package's quality and user experience. This includes data like feature usage frequency and system info, but never personal details. The data helps us prioritize improvements and ensure compatibility. If you wish to opt-out, set the environment variable `EC_TELEMETRY=false`. We prioritize data security and don't share this data externally.
|
||||
|
||||
## Citation
|
||||
|
||||
If you utilize this repository, please consider citing it with:
|
||||
|
||||
```
|
||||
@misc{embedchain,
|
||||
author = {Taranjeet Singh, Deshraj Yadav},
|
||||
title = {Embedchain: The Open Source RAG Framework},
|
||||
year = {2023},
|
||||
publisher = {GitHub},
|
||||
journal = {GitHub repository},
|
||||
howpublished = {\url{https://github.com/embedchain/embedchain}},
|
||||
}
|
||||
```
|
||||
@@ -1,10 +0,0 @@
|
||||
import importlib.metadata
|
||||
|
||||
__version__ = importlib.metadata.version(__package__ or __name__)
|
||||
|
||||
from embedchain.apps.App import App # noqa: F401
|
||||
from embedchain.apps.CustomApp import CustomApp # noqa: F401
|
||||
from embedchain.apps.Llama2App import Llama2App # noqa: F401
|
||||
from embedchain.apps.OpenSourceApp import OpenSourceApp # noqa: F401
|
||||
from embedchain.apps.PersonApp import (PersonApp, # noqa: F401
|
||||
PersonOpenSourceApp)
|
||||
@@ -1,63 +0,0 @@
|
||||
from typing import Optional
|
||||
|
||||
import openai
|
||||
|
||||
from embedchain.config import AppConfig, ChatConfig
|
||||
from embedchain.embedchain import EmbedChain
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class App(EmbedChain):
|
||||
"""
|
||||
The EmbedChain app.
|
||||
Has two functions: add and query.
|
||||
|
||||
adds(data_type, url): adds the data from the given URL to the vector db.
|
||||
query(query): finds answer to the given query using vector database and LLM.
|
||||
dry_run(query): test your prompt without consuming tokens.
|
||||
"""
|
||||
|
||||
def __init__(self, config: AppConfig = None, system_prompt: Optional[str] = None):
|
||||
"""
|
||||
:param config: AppConfig instance to load as configuration. Optional.
|
||||
:param system_prompt: System prompt string. Optional.
|
||||
"""
|
||||
if config is None:
|
||||
config = AppConfig()
|
||||
|
||||
super().__init__(config, system_prompt)
|
||||
|
||||
def get_llm_model_answer(self, prompt, config: ChatConfig):
|
||||
messages = []
|
||||
system_prompt = (
|
||||
self.system_prompt
|
||||
if self.system_prompt is not None
|
||||
else config.system_prompt
|
||||
if config.system_prompt is not None
|
||||
else None
|
||||
)
|
||||
if system_prompt:
|
||||
messages.append({"role": "system", "content": system_prompt})
|
||||
messages.append({"role": "user", "content": prompt})
|
||||
response = openai.ChatCompletion.create(
|
||||
model=config.model or "gpt-3.5-turbo-0613",
|
||||
messages=messages,
|
||||
temperature=config.temperature,
|
||||
max_tokens=config.max_tokens,
|
||||
top_p=config.top_p,
|
||||
stream=config.stream,
|
||||
)
|
||||
|
||||
if config.stream:
|
||||
return self._stream_llm_model_response(response)
|
||||
else:
|
||||
return response["choices"][0]["message"]["content"]
|
||||
|
||||
def _stream_llm_model_response(self, response):
|
||||
"""
|
||||
This is a generator for streaming response from the OpenAI completions API
|
||||
"""
|
||||
for line in response:
|
||||
chunk = line["choices"][0].get("delta", {}).get("content", "")
|
||||
yield chunk
|
||||
@@ -1,162 +0,0 @@
|
||||
import logging
|
||||
from typing import List, Optional
|
||||
|
||||
from langchain.schema import BaseMessage
|
||||
|
||||
from embedchain.config import ChatConfig, CustomAppConfig
|
||||
from embedchain.embedchain import EmbedChain
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
from embedchain.models import Providers
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class CustomApp(EmbedChain):
|
||||
"""
|
||||
The custom EmbedChain app.
|
||||
Has two functions: add and query.
|
||||
|
||||
adds(data_type, url): adds the data from the given URL to the vector db.
|
||||
query(query): finds answer to the given query using vector database and LLM.
|
||||
dry_run(query): test your prompt without consuming tokens.
|
||||
"""
|
||||
|
||||
def __init__(self, config: CustomAppConfig = None, system_prompt: Optional[str] = None):
|
||||
"""
|
||||
:param config: Optional. `CustomAppConfig` instance to load as configuration.
|
||||
:raises ValueError: Config must be provided for custom app
|
||||
:param system_prompt: Optional. System prompt string.
|
||||
"""
|
||||
if config is None:
|
||||
raise ValueError("Config must be provided for custom app")
|
||||
|
||||
self.provider = config.provider
|
||||
|
||||
if config.provider == Providers.GPT4ALL:
|
||||
from embedchain import OpenSourceApp
|
||||
|
||||
# Because these models run locally, they should have an instance running when the custom app is created
|
||||
self.open_source_app = OpenSourceApp(config=config.open_source_app_config)
|
||||
|
||||
super().__init__(config, system_prompt)
|
||||
|
||||
def set_llm_model(self, provider: Providers):
|
||||
self.provider = provider
|
||||
if provider == Providers.GPT4ALL:
|
||||
raise ValueError(
|
||||
"GPT4ALL needs to be instantiated with the model known, please create a new app instance instead"
|
||||
)
|
||||
|
||||
def get_llm_model_answer(self, prompt, config: ChatConfig):
|
||||
# TODO: Quitting the streaming response here for now.
|
||||
# Idea: https://gist.github.com/jvelezmagic/03ddf4c452d011aae36b2a0f73d72f68
|
||||
if config.stream:
|
||||
raise NotImplementedError(
|
||||
"Streaming responses have not been implemented for this model yet. Please disable."
|
||||
)
|
||||
|
||||
if config.system_prompt is None and self.system_prompt is not None:
|
||||
config.system_prompt = self.system_prompt
|
||||
|
||||
try:
|
||||
if self.provider == Providers.OPENAI:
|
||||
return CustomApp._get_openai_answer(prompt, config)
|
||||
|
||||
if self.provider == Providers.ANTHROPHIC:
|
||||
return CustomApp._get_athrophic_answer(prompt, config)
|
||||
|
||||
if self.provider == Providers.VERTEX_AI:
|
||||
return CustomApp._get_vertex_answer(prompt, config)
|
||||
|
||||
if self.provider == Providers.GPT4ALL:
|
||||
return self.open_source_app._get_gpt4all_answer(prompt, config)
|
||||
|
||||
if self.provider == Providers.AZURE_OPENAI:
|
||||
return CustomApp._get_azure_openai_answer(prompt, config)
|
||||
|
||||
except ImportError as e:
|
||||
raise ModuleNotFoundError(e.msg) from None
|
||||
|
||||
@staticmethod
|
||||
def _get_openai_answer(prompt: str, config: ChatConfig) -> str:
|
||||
from langchain.chat_models import ChatOpenAI
|
||||
|
||||
chat = ChatOpenAI(
|
||||
temperature=config.temperature,
|
||||
model=config.model or "gpt-3.5-turbo",
|
||||
max_tokens=config.max_tokens,
|
||||
streaming=config.stream,
|
||||
)
|
||||
|
||||
if config.top_p and config.top_p != 1:
|
||||
logging.warning("Config option `top_p` is not supported by this model.")
|
||||
|
||||
messages = CustomApp._get_messages(prompt, system_prompt=config.system_prompt)
|
||||
|
||||
return chat(messages).content
|
||||
|
||||
@staticmethod
|
||||
def _get_athrophic_answer(prompt: str, config: ChatConfig) -> str:
|
||||
from langchain.chat_models import ChatAnthropic
|
||||
|
||||
chat = ChatAnthropic(temperature=config.temperature, model=config.model)
|
||||
|
||||
if config.max_tokens and config.max_tokens != 1000:
|
||||
logging.warning("Config option `max_tokens` is not supported by this model.")
|
||||
|
||||
messages = CustomApp._get_messages(prompt, system_prompt=config.system_prompt)
|
||||
|
||||
return chat(messages).content
|
||||
|
||||
@staticmethod
|
||||
def _get_vertex_answer(prompt: str, config: ChatConfig) -> str:
|
||||
from langchain.chat_models import ChatVertexAI
|
||||
|
||||
chat = ChatVertexAI(temperature=config.temperature, model=config.model, max_output_tokens=config.max_tokens)
|
||||
|
||||
if config.top_p and config.top_p != 1:
|
||||
logging.warning("Config option `top_p` is not supported by this model.")
|
||||
|
||||
messages = CustomApp._get_messages(prompt, system_prompt=config.system_prompt)
|
||||
|
||||
return chat(messages).content
|
||||
|
||||
@staticmethod
|
||||
def _get_azure_openai_answer(prompt: str, config: ChatConfig) -> str:
|
||||
from langchain.chat_models import AzureChatOpenAI
|
||||
|
||||
if not config.deployment_name:
|
||||
raise ValueError("Deployment name must be provided for Azure OpenAI")
|
||||
|
||||
chat = AzureChatOpenAI(
|
||||
deployment_name=config.deployment_name,
|
||||
openai_api_version="2023-05-15",
|
||||
model_name=config.model or "gpt-3.5-turbo",
|
||||
temperature=config.temperature,
|
||||
max_tokens=config.max_tokens,
|
||||
streaming=config.stream,
|
||||
)
|
||||
|
||||
if config.top_p and config.top_p != 1:
|
||||
logging.warning("Config option `top_p` is not supported by this model.")
|
||||
|
||||
messages = CustomApp._get_messages(prompt, system_prompt=config.system_prompt)
|
||||
|
||||
return chat(messages).content
|
||||
|
||||
@staticmethod
|
||||
def _get_messages(prompt: str, system_prompt: Optional[str] = None) -> List[BaseMessage]:
|
||||
from langchain.schema import HumanMessage, SystemMessage
|
||||
|
||||
messages = []
|
||||
if system_prompt:
|
||||
messages.append(SystemMessage(content=system_prompt))
|
||||
messages.append(HumanMessage(content=prompt))
|
||||
return messages
|
||||
|
||||
def _stream_llm_model_response(self, response):
|
||||
"""
|
||||
This is a generator for streaming response from the OpenAI completions API
|
||||
"""
|
||||
for line in response:
|
||||
chunk = line["choices"][0].get("delta", {}).get("content", "")
|
||||
yield chunk
|
||||
@@ -1,40 +0,0 @@
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
from langchain.llms import Replicate
|
||||
|
||||
from embedchain.config import AppConfig, ChatConfig
|
||||
from embedchain.embedchain import EmbedChain
|
||||
|
||||
|
||||
class Llama2App(EmbedChain):
|
||||
"""
|
||||
The EmbedChain Llama2App class.
|
||||
Has two functions: add and query.
|
||||
|
||||
adds(data_type, url): adds the data from the given URL to the vector db.
|
||||
query(query): finds answer to the given query using vector database and LLM.
|
||||
"""
|
||||
|
||||
def __init__(self, config: AppConfig = None, system_prompt: Optional[str] = None):
|
||||
"""
|
||||
:param config: AppConfig instance to load as configuration. Optional.
|
||||
:param system_prompt: System prompt string. Optional.
|
||||
"""
|
||||
if "REPLICATE_API_TOKEN" not in os.environ:
|
||||
raise ValueError("Please set the REPLICATE_API_TOKEN environment variable.")
|
||||
|
||||
if config is None:
|
||||
config = AppConfig()
|
||||
|
||||
super().__init__(config, system_prompt)
|
||||
|
||||
def get_llm_model_answer(self, prompt, config: ChatConfig = None):
|
||||
# TODO: Move the model and other inputs into config
|
||||
if self.system_prompt or config.system_prompt:
|
||||
raise ValueError("Llama2App does not support `system_prompt`")
|
||||
llm = Replicate(
|
||||
model="a16z-infra/llama13b-v2-chat:df7690f1994d94e96ad9d568eac121aecf50684a0b0963b25a41cc40061269e5",
|
||||
input={"temperature": 0.75, "max_length": 500, "top_p": 1},
|
||||
)
|
||||
return llm(prompt)
|
||||
@@ -1,71 +0,0 @@
|
||||
import logging
|
||||
from typing import Iterable, Optional, Union
|
||||
|
||||
from embedchain.config import ChatConfig, OpenSourceAppConfig
|
||||
from embedchain.embedchain import EmbedChain
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
gpt4all_model = None
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class OpenSourceApp(EmbedChain):
|
||||
"""
|
||||
The OpenSource app.
|
||||
Same as App, but uses an open source embedding model and LLM.
|
||||
|
||||
Has two function: add and query.
|
||||
|
||||
adds(data_type, url): adds the data from the given URL to the vector db.
|
||||
query(query): finds answer to the given query using vector database and LLM.
|
||||
"""
|
||||
|
||||
def __init__(self, config: OpenSourceAppConfig = None, system_prompt: Optional[str] = None):
|
||||
"""
|
||||
:param config: OpenSourceAppConfig instance to load as configuration. Optional.
|
||||
`ef` defaults to open source.
|
||||
:param system_prompt: System prompt string. Optional.
|
||||
"""
|
||||
logging.info("Loading open source embedding model. This may take some time...") # noqa:E501
|
||||
if not config:
|
||||
config = OpenSourceAppConfig()
|
||||
|
||||
if not config.model:
|
||||
raise ValueError("OpenSourceApp needs a model to be instantiated. Maybe you passed the wrong config type?")
|
||||
|
||||
self.instance = OpenSourceApp._get_instance(config.model)
|
||||
|
||||
logging.info("Successfully loaded open source embedding model.")
|
||||
super().__init__(config, system_prompt)
|
||||
|
||||
def get_llm_model_answer(self, prompt, config: ChatConfig):
|
||||
return self._get_gpt4all_answer(prompt=prompt, config=config)
|
||||
|
||||
@staticmethod
|
||||
def _get_instance(model):
|
||||
try:
|
||||
from gpt4all import GPT4All
|
||||
except ModuleNotFoundError:
|
||||
raise ModuleNotFoundError(
|
||||
"The GPT4All python package is not installed. Please install it with `pip install embedchain[opensource]`" # noqa E501
|
||||
) from None
|
||||
|
||||
return GPT4All(model)
|
||||
|
||||
def _get_gpt4all_answer(self, prompt: str, config: ChatConfig) -> Union[str, Iterable]:
|
||||
if config.model and config.model != self.config.model:
|
||||
raise RuntimeError(
|
||||
"OpenSourceApp does not support switching models at runtime. Please create a new app instance."
|
||||
)
|
||||
|
||||
if self.system_prompt or config.system_prompt:
|
||||
raise ValueError("OpenSourceApp does not support `system_prompt`")
|
||||
|
||||
response = self.instance.generate(
|
||||
prompt=prompt,
|
||||
streaming=config.stream,
|
||||
top_p=config.top_p,
|
||||
max_tokens=config.max_tokens,
|
||||
temp=config.temperature,
|
||||
)
|
||||
return response
|
||||
@@ -1,83 +0,0 @@
|
||||
from string import Template
|
||||
|
||||
from embedchain.apps.App import App
|
||||
from embedchain.apps.OpenSourceApp import OpenSourceApp
|
||||
from embedchain.config import ChatConfig, QueryConfig
|
||||
from embedchain.config.apps.BaseAppConfig import BaseAppConfig
|
||||
from embedchain.config.QueryConfig import DEFAULT_PROMPT, DEFAULT_PROMPT_WITH_HISTORY
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class EmbedChainPersonApp:
|
||||
"""
|
||||
Base class to create a person bot.
|
||||
This bot behaves and speaks like a person.
|
||||
|
||||
:param person: name of the person, better if its a well known person.
|
||||
:param config: BaseAppConfig instance to load as configuration.
|
||||
"""
|
||||
|
||||
def __init__(self, person, config: BaseAppConfig = None):
|
||||
self.person = person
|
||||
self.person_prompt = f"You are {person}. Whatever you say, you will always say in {person} style." # noqa:E501
|
||||
super().__init__(config)
|
||||
|
||||
def add_person_template_to_config(self, default_prompt: str, config: ChatConfig = None):
|
||||
"""
|
||||
This method checks if the config object contains a prompt template
|
||||
if yes it adds the person prompt to it and return the updated config
|
||||
else it creates a config object with the default prompt added to the person prompt
|
||||
|
||||
:param default_prompt: it is the default prompt for query or chat methods
|
||||
:param config: Optional. The `ChatConfig` instance to use as
|
||||
configuration options.
|
||||
"""
|
||||
template = Template(self.person_prompt + " " + default_prompt)
|
||||
|
||||
if config:
|
||||
if config.template:
|
||||
# Add person prompt to custom user template
|
||||
config.template = Template(self.person_prompt + " " + config.template.template)
|
||||
else:
|
||||
# If no user template is present, use person prompt with the default template
|
||||
config.template = template
|
||||
else:
|
||||
# if no config is present at all, initialize the config with person prompt and default template
|
||||
config = QueryConfig(
|
||||
template=template,
|
||||
)
|
||||
|
||||
return config
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class PersonApp(EmbedChainPersonApp, App):
|
||||
"""
|
||||
The Person app.
|
||||
Extends functionality from EmbedChainPersonApp and App
|
||||
"""
|
||||
|
||||
def query(self, input_query, config: QueryConfig = None, dry_run=False):
|
||||
config = self.add_person_template_to_config(DEFAULT_PROMPT, config, where=None)
|
||||
return super().query(input_query, config, dry_run, where=None)
|
||||
|
||||
def chat(self, input_query, config: ChatConfig = None, dry_run=False, where=None):
|
||||
config = self.add_person_template_to_config(DEFAULT_PROMPT_WITH_HISTORY, config)
|
||||
return super().chat(input_query, config, dry_run, where)
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class PersonOpenSourceApp(EmbedChainPersonApp, OpenSourceApp):
|
||||
"""
|
||||
The Person app.
|
||||
Extends functionality from EmbedChainPersonApp and OpenSourceApp
|
||||
"""
|
||||
|
||||
def query(self, input_query, config: QueryConfig = None, dry_run=False):
|
||||
config = self.add_person_template_to_config(DEFAULT_PROMPT, config)
|
||||
return super().query(input_query, config, dry_run)
|
||||
|
||||
def chat(self, input_query, config: ChatConfig = None, dry_run=False):
|
||||
config = self.add_person_template_to_config(DEFAULT_PROMPT_WITH_HISTORY, config)
|
||||
return super().chat(input_query, config, dry_run)
|
||||
@@ -1,4 +0,0 @@
|
||||
from embedchain.bots.poe import PoeBot
|
||||
from embedchain.bots.whatsapp import WhatsAppBot
|
||||
# TODO: fix discord import
|
||||
# from embedchain.bots.discord import DiscordBot
|
||||
@@ -1,28 +0,0 @@
|
||||
from embedchain import CustomApp
|
||||
from embedchain.config import AddConfig, CustomAppConfig, QueryConfig
|
||||
from embedchain.helper_classes.json_serializable import (
|
||||
JSONSerializable, register_deserializable)
|
||||
from embedchain.models import EmbeddingFunctions, Providers
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class BaseBot(JSONSerializable):
|
||||
def __init__(self, app_config=None):
|
||||
if app_config is None:
|
||||
app_config = CustomAppConfig(embedding_fn=EmbeddingFunctions.OPENAI, provider=Providers.OPENAI)
|
||||
self.app_config = app_config
|
||||
self.app = CustomApp(config=self.app_config)
|
||||
|
||||
def add(self, data, config: AddConfig = None):
|
||||
"""Add data to the bot"""
|
||||
config = config if config else AddConfig()
|
||||
self.app.add(data, config=config)
|
||||
|
||||
def query(self, query, config: QueryConfig = None):
|
||||
"""Query bot"""
|
||||
config = config if config else QueryConfig()
|
||||
return self.app.query(query, config=config)
|
||||
|
||||
def start(self):
|
||||
"""Start the bot's functionality."""
|
||||
raise NotImplementedError("Subclasses must implement the start method.")
|
||||
@@ -1,64 +0,0 @@
|
||||
import hashlib
|
||||
|
||||
from embedchain.helper_classes.json_serializable import JSONSerializable
|
||||
from embedchain.models.data_type import DataType
|
||||
|
||||
|
||||
class BaseChunker(JSONSerializable):
|
||||
def __init__(self, text_splitter):
|
||||
"""Initialize the chunker."""
|
||||
self.text_splitter = text_splitter
|
||||
self.data_type = None
|
||||
|
||||
def create_chunks(self, loader, src):
|
||||
"""
|
||||
Loads data and chunks it.
|
||||
|
||||
:param loader: The loader which's `load_data` method is used to create
|
||||
the raw data.
|
||||
:param src: The data to be handled by the loader. Can be a URL for
|
||||
remote sources or local content for local loaders.
|
||||
"""
|
||||
documents = []
|
||||
ids = []
|
||||
idMap = {}
|
||||
datas = loader.load_data(src)
|
||||
metadatas = []
|
||||
for data in datas:
|
||||
content = data["content"]
|
||||
|
||||
meta_data = data["meta_data"]
|
||||
# add data type to meta data to allow query using data type
|
||||
meta_data["data_type"] = self.data_type.value
|
||||
url = meta_data["url"]
|
||||
|
||||
chunks = self.get_chunks(content)
|
||||
|
||||
for chunk in chunks:
|
||||
chunk_id = hashlib.sha256((chunk + url).encode()).hexdigest()
|
||||
if idMap.get(chunk_id) is None:
|
||||
idMap[chunk_id] = True
|
||||
ids.append(chunk_id)
|
||||
documents.append(chunk)
|
||||
metadatas.append(meta_data)
|
||||
return {
|
||||
"documents": documents,
|
||||
"ids": ids,
|
||||
"metadatas": metadatas,
|
||||
}
|
||||
|
||||
def get_chunks(self, content):
|
||||
"""
|
||||
Returns chunks using text splitter instance.
|
||||
|
||||
Override in child class if custom logic.
|
||||
"""
|
||||
return self.text_splitter.split_text(content)
|
||||
|
||||
def set_data_type(self, data_type: DataType):
|
||||
"""
|
||||
set the data type of chunker
|
||||
"""
|
||||
self.data_type = data_type
|
||||
|
||||
# TODO: This should be done during initialization. This means it has to be done in the child classes.
|
||||
@@ -1,46 +0,0 @@
|
||||
from typing import Callable, Optional
|
||||
|
||||
from embedchain.config.BaseConfig import BaseConfig
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class ChunkerConfig(BaseConfig):
|
||||
"""
|
||||
Config for the chunker used in `add` method
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
chunk_size: Optional[int] = None,
|
||||
chunk_overlap: Optional[int] = None,
|
||||
length_function: Optional[Callable[[str], int]] = None,
|
||||
):
|
||||
self.chunk_size = chunk_size if chunk_size else 2000
|
||||
self.chunk_overlap = chunk_overlap if chunk_overlap else 0
|
||||
self.length_function = length_function if length_function else len
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class LoaderConfig(BaseConfig):
|
||||
"""
|
||||
Config for the chunker used in `add` method
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
pass
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class AddConfig(BaseConfig):
|
||||
"""
|
||||
Config for the `add` method.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
chunker: Optional[ChunkerConfig] = None,
|
||||
loader: Optional[LoaderConfig] = None,
|
||||
):
|
||||
self.loader = loader
|
||||
self.chunker = chunker
|
||||
@@ -1,13 +0,0 @@
|
||||
from embedchain.helper_classes.json_serializable import JSONSerializable
|
||||
|
||||
|
||||
class BaseConfig(JSONSerializable):
|
||||
"""
|
||||
Base config.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
pass
|
||||
|
||||
def as_dict(self):
|
||||
return vars(self)
|
||||
@@ -1,92 +0,0 @@
|
||||
from string import Template
|
||||
from typing import Optional
|
||||
|
||||
from embedchain.config.QueryConfig import QueryConfig
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
DEFAULT_PROMPT = """
|
||||
You are a chatbot having a conversation with a human. You are given chat
|
||||
history and context.
|
||||
You need to answer the query considering context, chat history and your knowledge base. If you don't know the answer or the answer is neither contained in the context nor in history, then simply say "I don't know".
|
||||
|
||||
$context
|
||||
|
||||
History: $history
|
||||
|
||||
Query: $query
|
||||
|
||||
Helpful Answer:
|
||||
""" # noqa:E501
|
||||
|
||||
DEFAULT_PROMPT_TEMPLATE = Template(DEFAULT_PROMPT)
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class ChatConfig(QueryConfig):
|
||||
"""
|
||||
Config for the `chat` method, inherits from `QueryConfig`.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
number_documents=None,
|
||||
template: Template = None,
|
||||
model=None,
|
||||
temperature=None,
|
||||
max_tokens=None,
|
||||
top_p=None,
|
||||
stream: bool = False,
|
||||
deployment_name=None,
|
||||
system_prompt: Optional[str] = None,
|
||||
where=None,
|
||||
):
|
||||
"""
|
||||
Initializes the ChatConfig instance.
|
||||
|
||||
:param number_documents: Number of documents to pull from the database as
|
||||
context.
|
||||
:param template: Optional. The `Template` instance to use as a template for
|
||||
prompt.
|
||||
:param model: Optional. Controls the OpenAI model used.
|
||||
:param temperature: Optional. Controls the randomness of the model's output.
|
||||
Higher values (closer to 1) make output more random,lower values make it more
|
||||
deterministic.
|
||||
:param max_tokens: Optional. Controls how many tokens are generated.
|
||||
:param top_p: Optional. Controls the diversity of words.Higher values
|
||||
(closer to 1) make word selection more diverse, lower values make words less
|
||||
diverse.
|
||||
:param stream: Optional. Control if response is streamed back to the user
|
||||
:param deployment_name: t.b.a.
|
||||
:param system_prompt: Optional. System prompt string.
|
||||
:param where: Optional. A dictionary of key-value pairs to filter the database results.
|
||||
:raises ValueError: If the template is not valid as template should contain
|
||||
$context and $query and $history
|
||||
"""
|
||||
if template is None:
|
||||
template = DEFAULT_PROMPT_TEMPLATE
|
||||
|
||||
# History is set as 0 to ensure that there is always a history, that way,
|
||||
# there don't have to be two templates. Having two templates would make it
|
||||
# complicated because the history is not user controlled.
|
||||
super().__init__(
|
||||
number_documents=number_documents,
|
||||
template=template,
|
||||
model=model,
|
||||
temperature=temperature,
|
||||
max_tokens=max_tokens,
|
||||
top_p=top_p,
|
||||
history=[0],
|
||||
stream=stream,
|
||||
deployment_name=deployment_name,
|
||||
system_prompt=system_prompt,
|
||||
where=where,
|
||||
)
|
||||
|
||||
def set_history(self, history):
|
||||
"""
|
||||
Chat history is not user provided and not set at initialization time
|
||||
|
||||
:param history: (string) history to set
|
||||
"""
|
||||
self.history = history
|
||||
return
|
||||
@@ -1,148 +0,0 @@
|
||||
import re
|
||||
from string import Template
|
||||
from typing import Optional
|
||||
|
||||
from embedchain.config.BaseConfig import BaseConfig
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
DEFAULT_PROMPT = """
|
||||
Use the following pieces of context to answer the query at the end.
|
||||
If you don't know the answer, just say that you don't know, don't try to make up an answer.
|
||||
|
||||
$context
|
||||
|
||||
Query: $query
|
||||
|
||||
Helpful Answer:
|
||||
""" # noqa:E501
|
||||
|
||||
DEFAULT_PROMPT_WITH_HISTORY = """
|
||||
Use the following pieces of context to answer the query at the end.
|
||||
If you don't know the answer, just say that you don't know, don't try to make up an answer.
|
||||
I will provide you with our conversation history.
|
||||
|
||||
$context
|
||||
|
||||
History: $history
|
||||
|
||||
Query: $query
|
||||
|
||||
Helpful Answer:
|
||||
""" # noqa:E501
|
||||
|
||||
DOCS_SITE_DEFAULT_PROMPT = """
|
||||
Use the following pieces of context to answer the query at the end.
|
||||
If you don't know the answer, just say that you don't know, don't try to make up an answer. Wherever possible, give complete code snippet. Dont make up any code snippet on your own.
|
||||
|
||||
$context
|
||||
|
||||
Query: $query
|
||||
|
||||
Helpful Answer:
|
||||
""" # noqa:E501
|
||||
|
||||
DEFAULT_PROMPT_TEMPLATE = Template(DEFAULT_PROMPT)
|
||||
DEFAULT_PROMPT_WITH_HISTORY_TEMPLATE = Template(DEFAULT_PROMPT_WITH_HISTORY)
|
||||
DOCS_SITE_PROMPT_TEMPLATE = Template(DOCS_SITE_DEFAULT_PROMPT)
|
||||
query_re = re.compile(r"\$\{*query\}*")
|
||||
context_re = re.compile(r"\$\{*context\}*")
|
||||
history_re = re.compile(r"\$\{*history\}*")
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class QueryConfig(BaseConfig):
|
||||
"""
|
||||
Config for the `query` method.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
number_documents=None,
|
||||
template: Template = None,
|
||||
model=None,
|
||||
temperature=None,
|
||||
max_tokens=None,
|
||||
top_p=None,
|
||||
history=None,
|
||||
stream: bool = False,
|
||||
deployment_name=None,
|
||||
system_prompt: Optional[str] = None,
|
||||
where=None,
|
||||
):
|
||||
"""
|
||||
Initializes the QueryConfig instance.
|
||||
|
||||
:param number_documents: Number of documents to pull from the database as
|
||||
context.
|
||||
:param template: Optional. The `Template` instance to use as a template for
|
||||
prompt.
|
||||
:param model: Optional. Controls the OpenAI model used.
|
||||
:param temperature: Optional. Controls the randomness of the model's output.
|
||||
Higher values (closer to 1) make output more random, lower values make it more
|
||||
deterministic.
|
||||
:param max_tokens: Optional. Controls how many tokens are generated.
|
||||
:param top_p: Optional. Controls the diversity of words. Higher values
|
||||
(closer to 1) make word selection more diverse, lower values make words less
|
||||
diverse.
|
||||
:param history: Optional. A list of strings to consider as history.
|
||||
:param stream: Optional. Control if response is streamed back to user
|
||||
:param deployment_name: t.b.a.
|
||||
:param system_prompt: Optional. System prompt string.
|
||||
:param where: Optional. A dictionary of key-value pairs to filter the database results.
|
||||
:raises ValueError: If the template is not valid as template should
|
||||
contain $context and $query (and optionally $history).
|
||||
"""
|
||||
if number_documents is None:
|
||||
self.number_documents = 1
|
||||
else:
|
||||
self.number_documents = number_documents
|
||||
|
||||
if not history:
|
||||
self.history = None
|
||||
else:
|
||||
if len(history) == 0:
|
||||
self.history = None
|
||||
else:
|
||||
self.history = history
|
||||
|
||||
if template is None:
|
||||
if self.history is None:
|
||||
template = DEFAULT_PROMPT_TEMPLATE
|
||||
else:
|
||||
template = DEFAULT_PROMPT_WITH_HISTORY_TEMPLATE
|
||||
|
||||
self.temperature = temperature if temperature else 0
|
||||
self.max_tokens = max_tokens if max_tokens else 1000
|
||||
self.model = model
|
||||
self.top_p = top_p if top_p else 1
|
||||
self.deployment_name = deployment_name
|
||||
self.system_prompt = system_prompt
|
||||
|
||||
if self.validate_template(template):
|
||||
self.template = template
|
||||
else:
|
||||
if self.history is None:
|
||||
raise ValueError("`template` should have `query` and `context` keys")
|
||||
else:
|
||||
raise ValueError("`template` should have `query`, `context` and `history` keys")
|
||||
|
||||
if not isinstance(stream, bool):
|
||||
raise ValueError("`stream` should be bool")
|
||||
self.stream = stream
|
||||
self.where = where
|
||||
|
||||
def validate_template(self, template: Template):
|
||||
"""
|
||||
validate the template
|
||||
|
||||
:param template: the template to validate
|
||||
:return: Boolean, valid (true) or invalid (false)
|
||||
"""
|
||||
if self.history is None:
|
||||
return re.search(query_re, template.template) and re.search(context_re, template.template)
|
||||
else:
|
||||
return (
|
||||
re.search(query_re, template.template)
|
||||
and re.search(context_re, template.template)
|
||||
and re.search(history_re, template.template)
|
||||
)
|
||||
@@ -1,9 +0,0 @@
|
||||
from .AddConfig import AddConfig, ChunkerConfig # noqa: F401
|
||||
from .apps.AppConfig import AppConfig # noqa: F401
|
||||
from .apps.CustomAppConfig import CustomAppConfig # noqa: F401
|
||||
from .apps.OpenSourceAppConfig import OpenSourceAppConfig # noqa: F401
|
||||
from .BaseConfig import BaseConfig # noqa: F401
|
||||
from .ChatConfig import ChatConfig # noqa: F401
|
||||
from .QueryConfig import QueryConfig # noqa: F401
|
||||
from .vectordbs.ElasticsearchDBConfig import \
|
||||
ElasticsearchDBConfig # noqa: F401
|
||||
@@ -1,66 +0,0 @@
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
try:
|
||||
from chromadb.utils import embedding_functions
|
||||
except RuntimeError:
|
||||
from embedchain.utils import use_pysqlite3
|
||||
|
||||
use_pysqlite3()
|
||||
from chromadb.utils import embedding_functions
|
||||
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
from .BaseAppConfig import BaseAppConfig
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class AppConfig(BaseAppConfig):
|
||||
"""
|
||||
Config to initialize an embedchain custom `App` instance, with extra config options.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
log_level=None,
|
||||
host=None,
|
||||
port=None,
|
||||
id=None,
|
||||
collection_name=None,
|
||||
collect_metrics: Optional[bool] = None,
|
||||
):
|
||||
"""
|
||||
:param log_level: Optional. (String) Debug level
|
||||
['DEBUG', 'INFO', 'WARNING', 'ERROR', 'CRITICAL'].
|
||||
:param host: Optional. Hostname for the database server.
|
||||
:param port: Optional. Port for the database server.
|
||||
:param id: Optional. ID of the app. Document metadata will have this id.
|
||||
:param collection_name: Optional. Collection name for the database.
|
||||
:param collect_metrics: Defaults to True. Send anonymous telemetry to improve embedchain.
|
||||
"""
|
||||
super().__init__(
|
||||
log_level=log_level,
|
||||
embedding_fn=AppConfig.default_embedding_function(),
|
||||
host=host,
|
||||
port=port,
|
||||
id=id,
|
||||
collection_name=collection_name,
|
||||
collect_metrics=collect_metrics,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def default_embedding_function():
|
||||
"""
|
||||
Sets embedding function to default (`text-embedding-ada-002`).
|
||||
|
||||
:raises ValueError: If the template is not valid as template should contain
|
||||
$context and $query
|
||||
:returns: The default embedding function for the app class.
|
||||
"""
|
||||
if os.getenv("OPENAI_API_KEY") is None and os.getenv("OPENAI_ORGANIZATION") is None:
|
||||
raise ValueError("OPENAI_API_KEY or OPENAI_ORGANIZATION environment variables not provided") # noqa:E501
|
||||
return embedding_functions.OpenAIEmbeddingFunction(
|
||||
api_key=os.getenv("OPENAI_API_KEY"),
|
||||
organization_id=os.getenv("OPENAI_ORGANIZATION"),
|
||||
model_name="text-embedding-ada-002",
|
||||
)
|
||||
@@ -1,102 +0,0 @@
|
||||
import logging
|
||||
|
||||
from embedchain.config.BaseConfig import BaseConfig
|
||||
from embedchain.config.vectordbs import ElasticsearchDBConfig
|
||||
from embedchain.helper_classes.json_serializable import JSONSerializable
|
||||
from embedchain.models import VectorDatabases, VectorDimensions
|
||||
|
||||
|
||||
class BaseAppConfig(BaseConfig, JSONSerializable):
|
||||
"""
|
||||
Parent config to initialize an instance of `App`, `OpenSourceApp` or `CustomApp`.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
log_level=None,
|
||||
embedding_fn=None,
|
||||
db=None,
|
||||
host=None,
|
||||
port=None,
|
||||
id=None,
|
||||
collection_name=None,
|
||||
collect_metrics: bool = True,
|
||||
db_type: VectorDatabases = None,
|
||||
vector_dim: VectorDimensions = None,
|
||||
es_config: ElasticsearchDBConfig = None,
|
||||
chroma_settings: dict = {},
|
||||
):
|
||||
"""
|
||||
:param log_level: Optional. (String) Debug level
|
||||
['DEBUG', 'INFO', 'WARNING', 'ERROR', 'CRITICAL'].
|
||||
:param embedding_fn: Embedding function to use.
|
||||
:param db: Optional. (Vector) database instance to use for embeddings.
|
||||
:param host: Optional. Hostname for the database server.
|
||||
:param port: Optional. Port for the database server.
|
||||
:param id: Optional. ID of the app. Document metadata will have this id.
|
||||
:param collection_name: Optional. Collection name for the database.
|
||||
:param collect_metrics: Defaults to True. Send anonymous telemetry to improve embedchain.
|
||||
:param db_type: Optional. type of Vector database to use
|
||||
:param vector_dim: Vector dimension generated by embedding fn
|
||||
:param es_config: Optional. elasticsearch database config to be used for connection
|
||||
:param chroma_settings: Optional. Chroma settings for connection.
|
||||
"""
|
||||
self._setup_logging(log_level)
|
||||
self.collection_name = collection_name if collection_name else "embedchain_store"
|
||||
self.db = BaseAppConfig.get_db(
|
||||
db=db,
|
||||
embedding_fn=embedding_fn,
|
||||
host=host,
|
||||
port=port,
|
||||
db_type=db_type,
|
||||
vector_dim=vector_dim,
|
||||
collection_name=self.collection_name,
|
||||
es_config=es_config,
|
||||
chroma_settings=chroma_settings,
|
||||
)
|
||||
self.id = id
|
||||
self.collect_metrics = True if (collect_metrics is True or collect_metrics is None) else False
|
||||
return
|
||||
|
||||
@staticmethod
|
||||
def get_db(db, embedding_fn, host, port, db_type, vector_dim, collection_name, es_config, chroma_settings):
|
||||
"""
|
||||
Get db based on db_type, db with default database (`ChromaDb`)
|
||||
:param Optional. (Vector) database to use for embeddings.
|
||||
:param embedding_fn: Embedding function to use in database.
|
||||
:param host: Optional. Hostname for the database server.
|
||||
:param port: Optional. Port for the database server.
|
||||
:param db_type: Optional. db type to use. Supported values (`es`, `chroma`)
|
||||
:param vector_dim: Vector dimension generated by embedding fn
|
||||
:param collection_name: Optional. Collection name for the database.
|
||||
:param es_config: Optional. elasticsearch database config to be used for connection
|
||||
:raises ValueError: BaseAppConfig knows no default embedding function.
|
||||
:returns: database instance
|
||||
"""
|
||||
if db:
|
||||
return db
|
||||
|
||||
if embedding_fn is None:
|
||||
raise ValueError("ChromaDb cannot be instantiated without an embedding function")
|
||||
|
||||
if db_type == VectorDatabases.ELASTICSEARCH:
|
||||
from embedchain.vectordb.elasticsearch_db import ElasticsearchDB
|
||||
|
||||
return ElasticsearchDB(
|
||||
embedding_fn=embedding_fn, vector_dim=vector_dim, collection_name=collection_name, es_config=es_config
|
||||
)
|
||||
|
||||
from embedchain.vectordb.chroma_db import ChromaDB
|
||||
|
||||
return ChromaDB(embedding_fn=embedding_fn, host=host, port=port, chroma_settings=chroma_settings)
|
||||
|
||||
def _setup_logging(self, debug_level):
|
||||
level = logging.WARNING # Default level
|
||||
if debug_level is not None:
|
||||
level = getattr(logging, debug_level.upper(), None)
|
||||
if not isinstance(level, int):
|
||||
raise ValueError(f"Invalid log level: {debug_level}")
|
||||
|
||||
logging.basicConfig(format="%(asctime)s [%(name)s] [%(levelname)s] %(message)s", level=level)
|
||||
self.logger = logging.getLogger(__name__)
|
||||
return
|
||||
@@ -1,144 +0,0 @@
|
||||
from typing import Any, Optional
|
||||
|
||||
from chromadb.api.types import Documents, Embeddings
|
||||
from dotenv import load_dotenv
|
||||
|
||||
from embedchain.config.vectordbs import ElasticsearchDBConfig
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
from embedchain.models import (EmbeddingFunctions, Providers, VectorDatabases,
|
||||
VectorDimensions)
|
||||
|
||||
from .BaseAppConfig import BaseAppConfig
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class CustomAppConfig(BaseAppConfig):
|
||||
"""
|
||||
Config to initialize an embedchain custom `App` instance, with extra config options.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
log_level=None,
|
||||
embedding_fn: EmbeddingFunctions = None,
|
||||
embedding_fn_model=None,
|
||||
db=None,
|
||||
host=None,
|
||||
port=None,
|
||||
id=None,
|
||||
collection_name=None,
|
||||
provider: Providers = None,
|
||||
open_source_app_config=None,
|
||||
deployment_name=None,
|
||||
collect_metrics: Optional[bool] = None,
|
||||
db_type: VectorDatabases = None,
|
||||
es_config: ElasticsearchDBConfig = None,
|
||||
chroma_settings: dict = {},
|
||||
):
|
||||
"""
|
||||
:param log_level: Optional. (String) Debug level
|
||||
['DEBUG', 'INFO', 'WARNING', 'ERROR', 'CRITICAL'].
|
||||
:param embedding_fn: Optional. Embedding function to use.
|
||||
:param embedding_fn_model: Optional. Model name to use for embedding function.
|
||||
:param db: Optional. (Vector) database to use for embeddings.
|
||||
:param host: Optional. Hostname for the database server.
|
||||
:param port: Optional. Port for the database server.
|
||||
:param id: Optional. ID of the app. Document metadata will have this id.
|
||||
:param collection_name: Optional. Collection name for the database.
|
||||
:param provider: Optional. (Providers): LLM Provider to use.
|
||||
:param open_source_app_config: Optional. Config instance needed for open source apps.
|
||||
:param collect_metrics: Defaults to True. Send anonymous telemetry to improve embedchain.
|
||||
:param db_type: Optional. type of Vector database to use.
|
||||
:param es_config: Optional. elasticsearch database config to be used for connection
|
||||
:param chroma_settings: Optional. Chroma settings for connection.
|
||||
"""
|
||||
if provider:
|
||||
self.provider = provider
|
||||
else:
|
||||
raise ValueError("CustomApp must have a provider assigned.")
|
||||
|
||||
self.open_source_app_config = open_source_app_config
|
||||
|
||||
super().__init__(
|
||||
log_level=log_level,
|
||||
embedding_fn=CustomAppConfig.embedding_function(
|
||||
embedding_function=embedding_fn, model=embedding_fn_model, deployment_name=deployment_name
|
||||
),
|
||||
db=db,
|
||||
host=host,
|
||||
port=port,
|
||||
id=id,
|
||||
collection_name=collection_name,
|
||||
collect_metrics=collect_metrics,
|
||||
db_type=db_type,
|
||||
vector_dim=CustomAppConfig.get_vector_dimension(embedding_function=embedding_fn),
|
||||
es_config=es_config,
|
||||
chroma_settings=chroma_settings,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def langchain_default_concept(embeddings: Any):
|
||||
"""
|
||||
Langchains default function layout for embeddings.
|
||||
"""
|
||||
|
||||
def embed_function(texts: Documents) -> Embeddings:
|
||||
return embeddings.embed_documents(texts)
|
||||
|
||||
return embed_function
|
||||
|
||||
@staticmethod
|
||||
def embedding_function(embedding_function: EmbeddingFunctions, model: str = None, deployment_name: str = None):
|
||||
if not isinstance(embedding_function, EmbeddingFunctions):
|
||||
raise ValueError(
|
||||
f"Invalid option: '{embedding_function}'. Expecting one of the following options: {list(map(lambda x: x.value, EmbeddingFunctions))}" # noqa: E501
|
||||
)
|
||||
|
||||
if embedding_function == EmbeddingFunctions.OPENAI:
|
||||
from langchain.embeddings import OpenAIEmbeddings
|
||||
|
||||
if model:
|
||||
embeddings = OpenAIEmbeddings(model=model)
|
||||
else:
|
||||
if deployment_name:
|
||||
embeddings = OpenAIEmbeddings(deployment=deployment_name)
|
||||
else:
|
||||
embeddings = OpenAIEmbeddings()
|
||||
return CustomAppConfig.langchain_default_concept(embeddings)
|
||||
|
||||
elif embedding_function == EmbeddingFunctions.HUGGING_FACE:
|
||||
from langchain.embeddings import HuggingFaceEmbeddings
|
||||
|
||||
embeddings = HuggingFaceEmbeddings(model_name=model)
|
||||
return CustomAppConfig.langchain_default_concept(embeddings)
|
||||
|
||||
elif embedding_function == EmbeddingFunctions.VERTEX_AI:
|
||||
from langchain.embeddings import VertexAIEmbeddings
|
||||
|
||||
embeddings = VertexAIEmbeddings(model_name=model)
|
||||
return CustomAppConfig.langchain_default_concept(embeddings)
|
||||
|
||||
elif embedding_function == EmbeddingFunctions.GPT4ALL:
|
||||
# Note: We could use langchains GPT4ALL embedding, but it's not available in all versions.
|
||||
from chromadb.utils import embedding_functions
|
||||
|
||||
return embedding_functions.SentenceTransformerEmbeddingFunction(model_name=model)
|
||||
|
||||
@staticmethod
|
||||
def get_vector_dimension(embedding_function: EmbeddingFunctions):
|
||||
if not isinstance(embedding_function, EmbeddingFunctions):
|
||||
raise ValueError(f"Invalid option: '{embedding_function}'.")
|
||||
|
||||
if embedding_function == EmbeddingFunctions.OPENAI:
|
||||
return VectorDimensions.OPENAI.value
|
||||
|
||||
elif embedding_function == EmbeddingFunctions.HUGGING_FACE:
|
||||
return VectorDimensions.HUGGING_FACE.value
|
||||
|
||||
elif embedding_function == EmbeddingFunctions.VERTEX_AI:
|
||||
return VectorDimensions.VERTEX_AI.value
|
||||
|
||||
elif embedding_function == EmbeddingFunctions.GPT4ALL:
|
||||
return VectorDimensions.GPT4ALL.value
|
||||
@@ -1,62 +0,0 @@
|
||||
from typing import Optional
|
||||
|
||||
from chromadb.utils import embedding_functions
|
||||
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
from .BaseAppConfig import BaseAppConfig
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class OpenSourceAppConfig(BaseAppConfig):
|
||||
"""
|
||||
Config to initialize an embedchain custom `OpenSourceApp` instance, with extra config options.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
log_level=None,
|
||||
host=None,
|
||||
port=None,
|
||||
id=None,
|
||||
collection_name=None,
|
||||
collect_metrics: Optional[bool] = None,
|
||||
model=None,
|
||||
):
|
||||
"""
|
||||
:param log_level: Optional. (String) Debug level
|
||||
['DEBUG', 'INFO', 'WARNING', 'ERROR', 'CRITICAL'].
|
||||
:param id: Optional. ID of the app. Document metadata will have this id.
|
||||
:param collection_name: Optional. Collection name for the database.
|
||||
:param host: Optional. Hostname for the database server.
|
||||
:param port: Optional. Port for the database server.
|
||||
:param collect_metrics: Defaults to True. Send anonymous telemetry to improve embedchain.
|
||||
:param model: Optional. GPT4ALL uses the model to instantiate the class.
|
||||
So unlike `App`, it has to be provided before querying.
|
||||
"""
|
||||
self.model = model or "orca-mini-3b.ggmlv3.q4_0.bin"
|
||||
|
||||
super().__init__(
|
||||
log_level=log_level,
|
||||
embedding_fn=OpenSourceAppConfig.default_embedding_function(),
|
||||
host=host,
|
||||
port=port,
|
||||
id=id,
|
||||
collection_name=collection_name,
|
||||
collect_metrics=collect_metrics,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def default_embedding_function():
|
||||
"""
|
||||
Sets embedding function to default (`all-MiniLM-L6-v2`).
|
||||
|
||||
:returns: The default embedding function
|
||||
"""
|
||||
try:
|
||||
return embedding_functions.SentenceTransformerEmbeddingFunction(model_name="all-MiniLM-L6-v2")
|
||||
except ValueError as e:
|
||||
print(e)
|
||||
raise ModuleNotFoundError(
|
||||
"The open source app requires extra dependencies. Install with `pip install embedchain[opensource]`"
|
||||
) from None
|
||||
@@ -1,17 +0,0 @@
|
||||
from typing import Dict, List, Union
|
||||
|
||||
from embedchain.config.BaseConfig import BaseConfig
|
||||
from embedchain.helper_classes.json_serializable import register_deserializable
|
||||
|
||||
|
||||
@register_deserializable
|
||||
class ElasticsearchDBConfig(BaseConfig):
|
||||
"""
|
||||
Config to initialize an elasticsearch client.
|
||||
:param es_url. elasticsearch url or list of nodes url to be used for connection
|
||||
:param ES_EXTRA_PARAMS: extra params dict that can be passed to elasticsearch.
|
||||
"""
|
||||
|
||||
def __init__(self, es_url: Union[str, List[str]] = None, **ES_EXTRA_PARAMS: Dict[str, any]):
|
||||
self.ES_URL = es_url
|
||||
self.ES_EXTRA_PARAMS = ES_EXTRA_PARAMS
|
||||
@@ -0,0 +1,8 @@
|
||||
llm:
|
||||
provider: anthropic
|
||||
config:
|
||||
model: 'claude-instant-1'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
@@ -0,0 +1,19 @@
|
||||
app:
|
||||
config:
|
||||
id: azure-openai-app
|
||||
|
||||
llm:
|
||||
provider: azure_openai
|
||||
config:
|
||||
model: gpt-35-turbo
|
||||
deployment_name: your_llm_deployment_name
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
|
||||
embedder:
|
||||
provider: azure_openai
|
||||
config:
|
||||
model: text-embedding-ada-002
|
||||
deployment_name: you_embedding_model_deployment_name
|
||||
@@ -0,0 +1,24 @@
|
||||
app:
|
||||
config:
|
||||
id: 'my-app'
|
||||
|
||||
llm:
|
||||
provider: openai
|
||||
config:
|
||||
model: 'gpt-3.5-turbo'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
|
||||
vectordb:
|
||||
provider: chroma
|
||||
config:
|
||||
collection_name: 'my-app'
|
||||
dir: db
|
||||
allow_reset: true
|
||||
|
||||
embedder:
|
||||
provider: openai
|
||||
config:
|
||||
model: 'text-embedding-ada-002'
|
||||
@@ -0,0 +1,4 @@
|
||||
chunker:
|
||||
chunk_size: 100
|
||||
chunk_overlap: 20
|
||||
length_function: 'len'
|
||||
@@ -0,0 +1,12 @@
|
||||
llm:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/mistralai/completion/models/mistral-7B-Instruct"
|
||||
model_kwargs:
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
|
||||
embedder:
|
||||
provider: clarifai
|
||||
config:
|
||||
model: "https://clarifai.com/clarifai/main/models/BAAI-bge-base-en-v15"
|
||||
@@ -0,0 +1,7 @@
|
||||
llm:
|
||||
provider: cohere
|
||||
config:
|
||||
model: large
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
@@ -0,0 +1,40 @@
|
||||
app:
|
||||
config:
|
||||
id: 'full-stack-app'
|
||||
|
||||
chunker:
|
||||
chunk_size: 100
|
||||
chunk_overlap: 20
|
||||
length_function: 'len'
|
||||
|
||||
llm:
|
||||
provider: openai
|
||||
config:
|
||||
model: 'gpt-3.5-turbo'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
prompt: |
|
||||
Use the following pieces of context to answer the query at the end.
|
||||
If you don't know the answer, just say that you don't know, don't try to make up an answer.
|
||||
|
||||
$context
|
||||
|
||||
Query: $query
|
||||
|
||||
Helpful Answer:
|
||||
system_prompt: |
|
||||
Act as William Shakespeare. Answer the following questions in the style of William Shakespeare.
|
||||
|
||||
vectordb:
|
||||
provider: chroma
|
||||
config:
|
||||
collection_name: 'my-collection-name'
|
||||
dir: db
|
||||
allow_reset: true
|
||||
|
||||
embedder:
|
||||
provider: openai
|
||||
config:
|
||||
model: 'text-embedding-ada-002'
|
||||
@@ -0,0 +1,13 @@
|
||||
llm:
|
||||
provider: google
|
||||
config:
|
||||
model: gemini-pro
|
||||
max_tokens: 1000
|
||||
temperature: 0.9
|
||||
top_p: 1.0
|
||||
stream: false
|
||||
|
||||
embedder:
|
||||
provider: google
|
||||
config:
|
||||
model: models/embedding-001
|
||||
@@ -0,0 +1,8 @@
|
||||
llm:
|
||||
provider: openai
|
||||
config:
|
||||
model: 'gpt-4'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
@@ -0,0 +1,11 @@
|
||||
llm:
|
||||
provider: gpt4all
|
||||
config:
|
||||
model: 'orca-mini-3b-gguf2-q4_0.gguf'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
|
||||
embedder:
|
||||
provider: gpt4all
|
||||
@@ -0,0 +1,8 @@
|
||||
llm:
|
||||
provider: huggingface
|
||||
config:
|
||||
model: 'google/flan-t5-xxl'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 0.5
|
||||
stream: false
|
||||
@@ -0,0 +1,7 @@
|
||||
llm:
|
||||
provider: jina
|
||||
config:
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
@@ -0,0 +1,8 @@
|
||||
llm:
|
||||
provider: llama2
|
||||
config:
|
||||
model: 'a16z-infra/llama13b-v2-chat:df7690f1994d94e96ad9d568eac121aecf50684a0b0963b25a41cc40061269e5'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 0.5
|
||||
stream: false
|
||||
@@ -0,0 +1,14 @@
|
||||
llm:
|
||||
provider: ollama
|
||||
config:
|
||||
model: 'llama2'
|
||||
temperature: 0.5
|
||||
top_p: 1
|
||||
stream: true
|
||||
base_url: http://localhost:11434
|
||||
|
||||
embedder:
|
||||
provider: ollama
|
||||
config:
|
||||
model: 'mxbai-embed-large:latest'
|
||||
base_url: http://localhost:11434
|
||||
@@ -0,0 +1,33 @@
|
||||
app:
|
||||
config:
|
||||
id: 'my-app'
|
||||
log_level: 'WARNING'
|
||||
collect_metrics: true
|
||||
collection_name: 'my-app'
|
||||
|
||||
llm:
|
||||
provider: openai
|
||||
config:
|
||||
model: 'gpt-3.5-turbo'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
|
||||
vectordb:
|
||||
provider: opensearch
|
||||
config:
|
||||
opensearch_url: 'https://localhost:9200'
|
||||
http_auth:
|
||||
- admin
|
||||
- admin
|
||||
vector_dimension: 1536
|
||||
collection_name: 'my-app'
|
||||
use_ssl: false
|
||||
verify_certs: false
|
||||
|
||||
embedder:
|
||||
provider: openai
|
||||
config:
|
||||
model: 'text-embedding-ada-002'
|
||||
deployment_name: 'my-app'
|
||||
@@ -0,0 +1,25 @@
|
||||
app:
|
||||
config:
|
||||
id: 'open-source-app'
|
||||
collect_metrics: false
|
||||
|
||||
llm:
|
||||
provider: gpt4all
|
||||
config:
|
||||
model: 'orca-mini-3b-gguf2-q4_0.gguf'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
|
||||
vectordb:
|
||||
provider: chroma
|
||||
config:
|
||||
collection_name: 'open-source-app'
|
||||
dir: db
|
||||
allow_reset: true
|
||||
|
||||
embedder:
|
||||
provider: gpt4all
|
||||
config:
|
||||
deployment_name: 'test-deployment'
|
||||
@@ -0,0 +1,6 @@
|
||||
vectordb:
|
||||
provider: pinecone
|
||||
config:
|
||||
metric: cosine
|
||||
vector_dimension: 1536
|
||||
collection_name: my-pinecone-index
|
||||
@@ -0,0 +1,26 @@
|
||||
pipeline:
|
||||
config:
|
||||
name: Example pipeline
|
||||
id: pipeline-1 # Make sure that id is different every time you create a new pipeline
|
||||
|
||||
vectordb:
|
||||
provider: chroma
|
||||
config:
|
||||
collection_name: pipeline-1
|
||||
dir: db
|
||||
allow_reset: true
|
||||
|
||||
llm:
|
||||
provider: gpt4all
|
||||
config:
|
||||
model: 'orca-mini-3b-gguf2-q4_0.gguf'
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
top_p: 1
|
||||
stream: false
|
||||
|
||||
embedding_model:
|
||||
provider: gpt4all
|
||||
config:
|
||||
model: 'all-MiniLM-L6-v2'
|
||||
deployment_name: null
|
||||
@@ -0,0 +1,6 @@
|
||||
llm:
|
||||
provider: together
|
||||
config:
|
||||
model: mistralai/Mixtral-8x7B-Instruct-v0.1
|
||||
temperature: 0.5
|
||||
max_tokens: 1000
|
||||
@@ -0,0 +1,6 @@
|
||||
llm:
|
||||
provider: vertexai
|
||||
config:
|
||||
model: 'chat-bison'
|
||||
temperature: 0.5
|
||||
top_p: 0.5
|
||||
@@ -0,0 +1,14 @@
|
||||
llm:
|
||||
provider: vllm
|
||||
config:
|
||||
model: 'meta-llama/Llama-2-70b-hf'
|
||||
temperature: 0.5
|
||||
top_p: 1
|
||||
top_k: 10
|
||||
stream: true
|
||||
trust_remote_code: true
|
||||
|
||||
embedder:
|
||||
provider: huggingface
|
||||
config:
|
||||
model: 'BAAI/bge-small-en-v1.5'
|
||||
@@ -0,0 +1,4 @@
|
||||
vectordb:
|
||||
provider: weaviate
|
||||
config:
|
||||
collection_name: my_weaviate_index
|
||||
@@ -1,91 +0,0 @@
|
||||
from embedchain.chunkers.docs_site import DocsSiteChunker
|
||||
from embedchain.chunkers.docx_file import DocxFileChunker
|
||||
from embedchain.chunkers.notion import NotionChunker
|
||||
from embedchain.chunkers.pdf_file import PdfFileChunker
|
||||
from embedchain.chunkers.qna_pair import QnaPairChunker
|
||||
from embedchain.chunkers.text import TextChunker
|
||||
from embedchain.chunkers.web_page import WebPageChunker
|
||||
from embedchain.chunkers.youtube_video import YoutubeVideoChunker
|
||||
from embedchain.config import AddConfig
|
||||
from embedchain.helper_classes.json_serializable import JSONSerializable
|
||||
from embedchain.loaders.docs_site_loader import DocsSiteLoader
|
||||
from embedchain.loaders.docx_file import DocxFileLoader
|
||||
from embedchain.loaders.local_qna_pair import LocalQnaPairLoader
|
||||
from embedchain.loaders.local_text import LocalTextLoader
|
||||
from embedchain.loaders.pdf_file import PdfFileLoader
|
||||
from embedchain.loaders.sitemap import SitemapLoader
|
||||
from embedchain.loaders.web_page import WebPageLoader
|
||||
from embedchain.loaders.youtube_video import YoutubeVideoLoader
|
||||
from embedchain.models.data_type import DataType
|
||||
|
||||
|
||||
class DataFormatter(JSONSerializable):
|
||||
"""
|
||||
DataFormatter is an internal utility class which abstracts the mapping for
|
||||
loaders and chunkers to the data_type entered by the user in their
|
||||
.add or .add_local method call
|
||||
"""
|
||||
|
||||
def __init__(self, data_type: DataType, config: AddConfig):
|
||||
self.loader = self._get_loader(data_type, config.loader)
|
||||
self.chunker = self._get_chunker(data_type, config.chunker)
|
||||
|
||||
def _get_loader(self, data_type: DataType, config):
|
||||
"""
|
||||
Returns the appropriate data loader for the given data type.
|
||||
|
||||
:param data_type: The type of the data to load.
|
||||
:return: The loader for the given data type.
|
||||
:raises ValueError: If an unsupported data type is provided.
|
||||
"""
|
||||
loaders = {
|
||||
DataType.YOUTUBE_VIDEO: YoutubeVideoLoader,
|
||||
DataType.PDF_FILE: PdfFileLoader,
|
||||
DataType.WEB_PAGE: WebPageLoader,
|
||||
DataType.QNA_PAIR: LocalQnaPairLoader,
|
||||
DataType.TEXT: LocalTextLoader,
|
||||
DataType.DOCX: DocxFileLoader,
|
||||
DataType.SITEMAP: SitemapLoader,
|
||||
DataType.DOCS_SITE: DocsSiteLoader,
|
||||
}
|
||||
lazy_loaders = {DataType.NOTION}
|
||||
if data_type in loaders:
|
||||
loader_class = loaders[data_type]
|
||||
loader = loader_class()
|
||||
return loader
|
||||
elif data_type in lazy_loaders:
|
||||
if data_type == DataType.NOTION:
|
||||
from embedchain.loaders.notion import NotionLoader
|
||||
|
||||
return NotionLoader()
|
||||
else:
|
||||
raise ValueError(f"Unsupported data type: {data_type}")
|
||||
else:
|
||||
raise ValueError(f"Unsupported data type: {data_type}")
|
||||
|
||||
def _get_chunker(self, data_type: DataType, config):
|
||||
"""
|
||||
Returns the appropriate chunker for the given data type.
|
||||
|
||||
:param data_type: The type of the data to chunk.
|
||||
:return: The chunker for the given data type.
|
||||
:raises ValueError: If an unsupported data type is provided.
|
||||
"""
|
||||
chunker_classes = {
|
||||
DataType.YOUTUBE_VIDEO: YoutubeVideoChunker,
|
||||
DataType.PDF_FILE: PdfFileChunker,
|
||||
DataType.WEB_PAGE: WebPageChunker,
|
||||
DataType.QNA_PAIR: QnaPairChunker,
|
||||
DataType.TEXT: TextChunker,
|
||||
DataType.DOCX: DocxFileChunker,
|
||||
DataType.WEB_PAGE: WebPageChunker,
|
||||
DataType.DOCS_SITE: DocsSiteChunker,
|
||||
DataType.NOTION: NotionChunker,
|
||||
}
|
||||
if data_type in chunker_classes:
|
||||
chunker_class = chunker_classes[data_type]
|
||||
chunker = chunker_class(config)
|
||||
chunker.set_data_type(data_type)
|
||||
return chunker
|
||||
else:
|
||||
raise ValueError(f"Unsupported data type: {data_type}")
|
||||
@@ -0,0 +1,10 @@
|
||||
install:
|
||||
npm i -g mintlify
|
||||
|
||||
run_local:
|
||||
mintlify dev
|
||||
|
||||
troubleshoot:
|
||||
mintlify install
|
||||
|
||||
.PHONY: install run_local troubleshoot
|
||||
@@ -0,0 +1,25 @@
|
||||
# Contributing to embedchain docs
|
||||
|
||||
|
||||
### 👩💻 Development
|
||||
|
||||
Install the [Mintlify CLI](https://www.npmjs.com/package/mintlify) to preview the documentation changes locally. To install, use the following command
|
||||
|
||||
```
|
||||
npm i -g mintlify
|
||||
```
|
||||
|
||||
Run the following command at the root of your documentation (where mint.json is)
|
||||
|
||||
```
|
||||
mintlify dev
|
||||
```
|
||||
|
||||
### 😎 Publishing Changes
|
||||
|
||||
Changes will be deployed to production automatically after your PR is merged to the main branch.
|
||||
|
||||
#### Troubleshooting
|
||||
|
||||
- Mintlify dev isn't running - Run `mintlify install` it'll re-install dependencies.
|
||||
- Page loads as a 404 - Make sure you are running in a folder with `mint.json`
|
||||
@@ -0,0 +1,11 @@
|
||||
<CardGroup cols={3}>
|
||||
<Card title="Talk to founders" icon="calendar" href="https://cal.com/taranjeetio/ec">
|
||||
Schedule a call
|
||||
</Card>
|
||||
<Card title="Slack" icon="slack" href="https://embedchain.ai/slack" color="#4A154B">
|
||||
Join our slack community
|
||||
</Card>
|
||||
<Card title="Discord" icon="discord" href="https://discord.gg/6PzXDgEjG5" color="#7289DA">
|
||||
Join our discord community
|
||||
</Card>
|
||||
</CardGroup>
|
||||