json.py 913 B

1234567891011121314151617181920212223
  1. import hashlib
  2. from langchain.document_loaders.json_loader import JSONLoader as LcJSONLoader
  3. from embedchain.loaders.base_loader import BaseLoader
  4. langchain_json_jq_schema = 'to_entries | map("\(.key): \(.value|tostring)") | .[]'
  5. class JSONLoader(BaseLoader):
  6. @staticmethod
  7. def load_data(content):
  8. """Load a json file. Each data point is a key value pair."""
  9. data = []
  10. data_content = []
  11. loader = LcJSONLoader(content, text_content=False, jq_schema=langchain_json_jq_schema)
  12. docs = loader.load()
  13. for doc in docs:
  14. meta_data = doc.metadata
  15. data.append({"content": doc.page_content, "meta_data": {"url": content, "row": meta_data["seq_num"]}})
  16. data_content.append(doc.page_content)
  17. doc_id = hashlib.sha256((content + ", ".join(data_content)).encode()).hexdigest()
  18. return {"doc_id": doc_id, "data": data}