Embedchain json url support (#878)

Co-authored-by: Deven Patel <deven298@yahoo.com>
This commit is contained in:
Deven Patel
2023-10-30 16:19:11 -07:00
committed by GitHub
parent 68dc274f72
commit 5255a37c93
2 changed files with 84 additions and 8 deletions

View File

@@ -1,9 +1,14 @@
import hashlib
import json
import os
import re
import requests
from embedchain.loaders.base_loader import BaseLoader
VALID_URL_PATTERN = "^https:\/\/[0-9A-z.]+.[0-9A-z.]+.[a-z]+\/.*\.json$"
class JSONLoader(BaseLoader):
@staticmethod
@@ -20,18 +25,32 @@ class JSONLoader(BaseLoader):
loader = LLHBUBJSONLoader()
if not isinstance(content, str) and not os.path.isfile(content):
if not isinstance(content, str):
print(f"Invaid content input. Provide the correct path to the json file saved locally in {content}")
data = []
data_content = []
with open(content, "r") as json_file:
json_data = json.load(json_file)
docs = loader.load_data(json_data)
for doc in docs:
doc_content = doc.text
data.append({"content": doc_content, "meta_data": {"url": content}})
data_content.append(doc_content)
# Load json data from various sources. TODO: add support for dictionary
if os.path.isfile(content):
with open(content, "r") as json_file:
json_data = json.load(json_file)
elif re.match(VALID_URL_PATTERN, content):
response = requests.get(content)
if response.status_code == 200:
json_data = response.json()
else:
raise ValueError(
f"Loading data from the given url: {content} failed. \
Make sure the url is working."
)
else:
raise ValueError(f"Invalid content to load json data from: {content}")
docs = loader.load_data(json_data)
for doc in docs:
doc_content = doc.text
data.append({"content": doc_content, "meta_data": {"url": content}})
data_content.append(doc_content)
doc_id = hashlib.sha256((content + ", ".join(data_content)).encode()).hexdigest()
return {"doc_id": doc_id, "data": data}