feat: add document loaders
This commit is contained in:
parent
296ecafdcf
commit
c69e5a72f0
15 changed files with 1327 additions and 125 deletions
9
.gitignore
vendored
9
.gitignore
vendored
|
|
@ -9,6 +9,11 @@ lerna-debug.log*
|
||||||
# Mac
|
# Mac
|
||||||
.DS_Store
|
.DS_Store
|
||||||
|
|
||||||
|
# VSCode
|
||||||
|
.vscode
|
||||||
|
.chroma
|
||||||
|
.ruff_cache
|
||||||
|
|
||||||
# Diagnostic reports (https://nodejs.org/api/report.html)
|
# Diagnostic reports (https://nodejs.org/api/report.html)
|
||||||
report.[0-9]*.[0-9]*.[0-9]*.[0-9]*.json
|
report.[0-9]*.[0-9]*.[0-9]*.[0-9]*.json
|
||||||
|
|
||||||
|
|
@ -233,5 +238,5 @@ venv.bak/
|
||||||
.dmypy.json
|
.dmypy.json
|
||||||
dmypy.json
|
dmypy.json
|
||||||
|
|
||||||
# Poetry
|
# Poetry
|
||||||
.testenv/*
|
.testenv/*
|
||||||
|
|
|
||||||
1098
poetry.lock
generated
1098
poetry.lock
generated
File diff suppressed because it is too large
Load diff
|
|
@ -39,6 +39,12 @@ huggingface-hub = "^0.13.3"
|
||||||
rich = "^13.3.3"
|
rich = "^13.3.3"
|
||||||
llama-cpp-python = "0.1.23"
|
llama-cpp-python = "0.1.23"
|
||||||
networkx = "^3.1"
|
networkx = "^3.1"
|
||||||
|
unstructured = "^0.5.11"
|
||||||
|
pypdf = "^3.7.1"
|
||||||
|
lxml = "^4.9.2"
|
||||||
|
unstructured-inference = "^0.3.2"
|
||||||
|
pysrt = "^1.1.2"
|
||||||
|
fake-useragent = "^1.1.3"
|
||||||
|
|
||||||
[tool.poetry.group.dev.dependencies]
|
[tool.poetry.group.dev.dependencies]
|
||||||
black = "^23.1.0"
|
black = "^23.1.0"
|
||||||
|
|
|
||||||
34
src/backend/langflow/cache/utils.py
vendored
34
src/backend/langflow/cache/utils.py
vendored
|
|
@ -1,3 +1,4 @@
|
||||||
|
import base64
|
||||||
import contextlib
|
import contextlib
|
||||||
import functools
|
import functools
|
||||||
import hashlib
|
import hashlib
|
||||||
|
|
@ -84,6 +85,39 @@ def filter_json(json_data):
|
||||||
return filtered_data
|
return filtered_data
|
||||||
|
|
||||||
|
|
||||||
|
def save_binary_file(content: str, file_name: str, accepted_types: list[str]) -> str:
|
||||||
|
"""
|
||||||
|
Save a binary file to the specified folder.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
content: The content of the file as a bytes object.
|
||||||
|
file_name: The name of the file, including its extension.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The path to the saved file.
|
||||||
|
"""
|
||||||
|
if not any(file_name.endswith(suffix) for suffix in accepted_types):
|
||||||
|
raise ValueError(f"File {file_name} is not accepted")
|
||||||
|
|
||||||
|
# Get the destination folder
|
||||||
|
cache_path = Path(tempfile.gettempdir()) / PREFIX
|
||||||
|
|
||||||
|
data = content.split(",")[1]
|
||||||
|
decoded_bytes = base64.b64decode(data)
|
||||||
|
|
||||||
|
# Create the destination folder if it doesn't exist
|
||||||
|
os.makedirs(cache_path, exist_ok=True)
|
||||||
|
|
||||||
|
# Create the full file path
|
||||||
|
file_path = os.path.join(cache_path, file_name)
|
||||||
|
|
||||||
|
# Save the binary content to the file
|
||||||
|
with open(file_path, "wb") as file:
|
||||||
|
file.write(decoded_bytes)
|
||||||
|
|
||||||
|
return file_path
|
||||||
|
|
||||||
|
|
||||||
def save_cache(hash_val: str, chat_data, clean_old_cache_files: bool):
|
def save_cache(hash_val: str, chat_data, clean_old_cache_files: bool):
|
||||||
cache_path = Path(tempfile.gettempdir()) / PREFIX / f"{hash_val}.dill"
|
cache_path = Path(tempfile.gettempdir()) / PREFIX / f"{hash_val}.dill"
|
||||||
with cache_path.open("wb") as cache_file:
|
with cache_path.open("wb") as cache_file:
|
||||||
|
|
|
||||||
|
|
@ -61,8 +61,31 @@ vectorstores:
|
||||||
- Chroma
|
- Chroma
|
||||||
|
|
||||||
documentloaders:
|
documentloaders:
|
||||||
|
- AirbyteJSONLoader
|
||||||
|
- CoNLLULoader
|
||||||
|
- CSVLoader
|
||||||
|
- UnstructuredEmailLoader
|
||||||
|
- EverNoteLoader
|
||||||
|
- FacebookChatLoader
|
||||||
|
- GutenbergLoader
|
||||||
|
- BSHTMLLoader
|
||||||
|
- UnstructuredHTMLLoader
|
||||||
|
- UnstructuredImageLoader
|
||||||
|
- UnstructuredMarkdownLoader
|
||||||
|
- PyPDFLoader
|
||||||
|
- UnstructuredPowerPointLoader
|
||||||
|
- SRTLoader
|
||||||
|
- TelegramChatLoader
|
||||||
- TextLoader
|
- TextLoader
|
||||||
|
- UnstructuredWordDocumentLoader
|
||||||
- WebBaseLoader
|
- WebBaseLoader
|
||||||
|
- AZLyricsLoader
|
||||||
|
- CollegeConfidentialLoader
|
||||||
|
- HNLoader
|
||||||
|
- IFixitLoader
|
||||||
|
- IMSDbLoader
|
||||||
|
- GitbookLoader
|
||||||
|
- ReadTheDocsLoader
|
||||||
|
|
||||||
textsplitters:
|
textsplitters:
|
||||||
- CharacterTextSplitter
|
- CharacterTextSplitter
|
||||||
|
|
|
||||||
|
|
@ -8,8 +8,8 @@ import warnings
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from typing import Any, Dict, List, Optional
|
from typing import Any, Dict, List, Optional
|
||||||
|
|
||||||
|
from langflow.cache import utils as cache_utils
|
||||||
from langflow.graph.constants import DIRECT_TYPES
|
from langflow.graph.constants import DIRECT_TYPES
|
||||||
from langflow.graph.utils import load_file
|
|
||||||
from langflow.interface import loading
|
from langflow.interface import loading
|
||||||
from langflow.interface.listing import ALL_TYPES_DICT
|
from langflow.interface.listing import ALL_TYPES_DICT
|
||||||
from langflow.utils.logger import logger
|
from langflow.utils.logger import logger
|
||||||
|
|
@ -88,8 +88,11 @@ class Node:
|
||||||
file_name = value.get("value")
|
file_name = value.get("value")
|
||||||
content = value.get("content")
|
content = value.get("content")
|
||||||
type_to_load = value.get("suffixes")
|
type_to_load = value.get("suffixes")
|
||||||
loaded_dict = load_file(file_name, content, type_to_load)
|
file_path = cache_utils.save_binary_file(
|
||||||
params[key] = loaded_dict
|
content=content, file_name=file_name, accepted_types=type_to_load
|
||||||
|
)
|
||||||
|
|
||||||
|
params[key] = file_path
|
||||||
|
|
||||||
# We should check if the type is in something not
|
# We should check if the type is in something not
|
||||||
# the opposite
|
# the opposite
|
||||||
|
|
|
||||||
|
|
@ -1,48 +1,4 @@
|
||||||
import base64
|
|
||||||
import csv
|
|
||||||
import io
|
|
||||||
import json
|
|
||||||
import re
|
import re
|
||||||
from typing import Any
|
|
||||||
|
|
||||||
import yaml
|
|
||||||
|
|
||||||
|
|
||||||
def load_file(file_name, file_content, accepted_types) -> Any:
|
|
||||||
"""Load a file from a string."""
|
|
||||||
# Check if the file is accepted
|
|
||||||
if not any(file_name.endswith(suffix) for suffix in accepted_types):
|
|
||||||
raise ValueError(f"File {file_name} is not accepted")
|
|
||||||
# Get the suffix
|
|
||||||
suffix = file_name.split(".")[-1]
|
|
||||||
# file_content == 'data:application/x-yaml;base64,b3BlbmFwaTogIjMuMC4wIg...'
|
|
||||||
data = file_content.split(",")[1]
|
|
||||||
decoded_bytes = base64.b64decode(data)
|
|
||||||
|
|
||||||
# Convert the bytes object to a string
|
|
||||||
decoded_string = decoded_bytes.decode("utf-8")
|
|
||||||
if suffix == "json":
|
|
||||||
# Return the json content
|
|
||||||
return json.loads(decoded_string)
|
|
||||||
elif suffix in ["yaml", "yml"]:
|
|
||||||
# Return the yaml content
|
|
||||||
loaded_yaml = yaml.load(decoded_string, Loader=yaml.FullLoader)
|
|
||||||
try:
|
|
||||||
from langchain.agents.agent_toolkits.openapi.spec import reduce_openapi_spec # type: ignore
|
|
||||||
|
|
||||||
return reduce_openapi_spec(loaded_yaml)
|
|
||||||
except ImportError:
|
|
||||||
return loaded_yaml
|
|
||||||
|
|
||||||
elif suffix == "csv":
|
|
||||||
# Load the csv content
|
|
||||||
csv_reader = csv.DictReader(io.StringIO(decoded_string))
|
|
||||||
return list(csv_reader)
|
|
||||||
elif suffix == "txt":
|
|
||||||
# Return the text content
|
|
||||||
return decoded_string
|
|
||||||
else:
|
|
||||||
raise ValueError(f"File {file_name} is not accepted")
|
|
||||||
|
|
||||||
|
|
||||||
def validate_prompt(prompt: str):
|
def validate_prompt(prompt: str):
|
||||||
|
|
|
||||||
|
|
@ -77,7 +77,7 @@ class CSVAgent(AgentExecutor):
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_toolkit_and_llm(
|
def from_toolkit_and_llm(
|
||||||
cls,
|
cls,
|
||||||
path: dict,
|
path: str,
|
||||||
llm: BaseLanguageModel,
|
llm: BaseLanguageModel,
|
||||||
pandas_kwargs: Optional[dict] = None,
|
pandas_kwargs: Optional[dict] = None,
|
||||||
**kwargs: Any
|
**kwargs: Any
|
||||||
|
|
@ -85,7 +85,7 @@ class CSVAgent(AgentExecutor):
|
||||||
import pandas as pd # type: ignore
|
import pandas as pd # type: ignore
|
||||||
|
|
||||||
_kwargs = pandas_kwargs or {}
|
_kwargs = pandas_kwargs or {}
|
||||||
df = pd.DataFrame.from_dict(path, **_kwargs)
|
df = pd.read_csv(path, **_kwargs)
|
||||||
|
|
||||||
tools = [PythonAstREPLTool(locals={"df": df})] # type: ignore
|
tools = [PythonAstREPLTool(locals={"df": df})] # type: ignore
|
||||||
prompt = ZeroShotAgent.create_prompt(
|
prompt = ZeroShotAgent.create_prompt(
|
||||||
|
|
|
||||||
|
|
@ -2,25 +2,31 @@ from typing import Dict, List, Optional
|
||||||
|
|
||||||
from langflow.interface.base import LangChainTypeCreator
|
from langflow.interface.base import LangChainTypeCreator
|
||||||
from langflow.interface.custom_lists import documentloaders_type_to_cls_dict
|
from langflow.interface.custom_lists import documentloaders_type_to_cls_dict
|
||||||
from langflow.interface.documentLoaders.custom import CUSTOM_DOCUMENTLOADERS
|
|
||||||
from langflow.settings import settings
|
from langflow.settings import settings
|
||||||
from langflow.utils.util import build_template_from_class
|
from langflow.utils.util import build_template_from_class
|
||||||
|
|
||||||
|
|
||||||
|
def build_file_path_template(
|
||||||
|
suffixes: list, fileTypes: list, name: str = "file_path"
|
||||||
|
) -> Dict:
|
||||||
|
"""Build a file path template for a document loader."""
|
||||||
|
return {
|
||||||
|
"type": "file",
|
||||||
|
"required": True,
|
||||||
|
"show": True,
|
||||||
|
"name": name,
|
||||||
|
"value": "",
|
||||||
|
"suffixes": suffixes,
|
||||||
|
"fileTypes": fileTypes,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
class DocumentLoaderCreator(LangChainTypeCreator):
|
class DocumentLoaderCreator(LangChainTypeCreator):
|
||||||
type_name: str = "documentloaders"
|
type_name: str = "documentloaders"
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def type_to_loader_dict(self) -> Dict:
|
def type_to_loader_dict(self) -> Dict:
|
||||||
types = documentloaders_type_to_cls_dict
|
return documentloaders_type_to_cls_dict
|
||||||
|
|
||||||
# Drop some types that are reimplemented with the same name
|
|
||||||
types.pop("TextLoader")
|
|
||||||
|
|
||||||
for name, documentloader in CUSTOM_DOCUMENTLOADERS.items():
|
|
||||||
types[name] = documentloader
|
|
||||||
|
|
||||||
return types
|
|
||||||
|
|
||||||
def get_signature(self, name: str) -> Optional[Dict]:
|
def get_signature(self, name: str) -> Optional[Dict]:
|
||||||
"""Get the signature of a document loader."""
|
"""Get the signature of a document loader."""
|
||||||
|
|
@ -29,29 +35,101 @@ class DocumentLoaderCreator(LangChainTypeCreator):
|
||||||
name, documentloaders_type_to_cls_dict
|
name, documentloaders_type_to_cls_dict
|
||||||
)
|
)
|
||||||
|
|
||||||
if name == "TextLoader":
|
file_path_templates = {
|
||||||
signature["template"]["file"] = {
|
"AirbyteJSONLoader": build_file_path_template(
|
||||||
"type": "file",
|
suffixes=[".json"], fileTypes=["json"]
|
||||||
"required": True,
|
),
|
||||||
"show": True,
|
"CoNLLULoader": build_file_path_template(
|
||||||
"name": "path",
|
suffixes=[".csv"], fileTypes=["csv"]
|
||||||
"value": "",
|
),
|
||||||
"suffixes": [".txt"],
|
"CSVLoader": build_file_path_template(
|
||||||
"fileTypes": ["txt"],
|
suffixes=[".csv"], fileTypes=["csv"]
|
||||||
}
|
),
|
||||||
elif name == "WebBaseLoader":
|
"UnstructuredEmailLoader": build_file_path_template(
|
||||||
|
suffixes=[".eml"], fileTypes=["eml"]
|
||||||
|
),
|
||||||
|
"EverNoteLoader": build_file_path_template(
|
||||||
|
suffixes=[".xml"], fileTypes=["xml"]
|
||||||
|
),
|
||||||
|
"FacebookChatLoader": build_file_path_template(
|
||||||
|
suffixes=[".json"], fileTypes=["json"]
|
||||||
|
),
|
||||||
|
"GutenbergLoader": build_file_path_template(
|
||||||
|
suffixes=[".txt"], fileTypes=["txt"]
|
||||||
|
),
|
||||||
|
"BSHTMLLoader": build_file_path_template(
|
||||||
|
suffixes=[".html"], fileTypes=["html"]
|
||||||
|
),
|
||||||
|
"UnstructuredHTMLLoader": build_file_path_template(
|
||||||
|
suffixes=[".html"], fileTypes=["html"]
|
||||||
|
),
|
||||||
|
"UnstructuredImageLoader": build_file_path_template(
|
||||||
|
suffixes=[".jpg", ".jpeg", ".png", ".gif", ".bmp"],
|
||||||
|
fileTypes=["jpg", "jpeg", "png", "gif", "bmp"],
|
||||||
|
),
|
||||||
|
"UnstructuredMarkdownLoader": build_file_path_template(
|
||||||
|
suffixes=[".md"], fileTypes=["md"]
|
||||||
|
),
|
||||||
|
"PyPDFLoader": build_file_path_template(
|
||||||
|
suffixes=[".pdf"], fileTypes=["pdf"]
|
||||||
|
),
|
||||||
|
"UnstructuredPowerPointLoader": build_file_path_template(
|
||||||
|
suffixes=[".pptx", ".ppt"], fileTypes=["pptx", "ppt"]
|
||||||
|
),
|
||||||
|
"SRTLoader": build_file_path_template(
|
||||||
|
suffixes=[".srt"], fileTypes=["srt"]
|
||||||
|
),
|
||||||
|
"TelegramChatLoader": build_file_path_template(
|
||||||
|
suffixes=[".json"], fileTypes=["json"]
|
||||||
|
),
|
||||||
|
"TextLoader": build_file_path_template(
|
||||||
|
suffixes=[".txt"], fileTypes=["txt"]
|
||||||
|
),
|
||||||
|
"UnstructuredWordDocumentLoader": build_file_path_template(
|
||||||
|
suffixes=[".docx", ".doc"], fileTypes=["docx", "doc"]
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
if name in file_path_templates:
|
||||||
|
signature["template"]["file_path"] = file_path_templates[name]
|
||||||
|
elif name in {
|
||||||
|
"WebBaseLoader",
|
||||||
|
"AZLyricsLoader",
|
||||||
|
"CollegeConfidentialLoader",
|
||||||
|
"HNLoader",
|
||||||
|
"IFixitLoader",
|
||||||
|
"IMSDbLoader",
|
||||||
|
}:
|
||||||
signature["template"]["web_path"] = {
|
signature["template"]["web_path"] = {
|
||||||
"type": "str",
|
"type": "str",
|
||||||
"required": True,
|
"required": True,
|
||||||
"show": True,
|
"show": True,
|
||||||
"name": "web_path",
|
"name": "web_path",
|
||||||
"value": "",
|
"value": "",
|
||||||
"display_name": "Web Path",
|
"display_name": "Web Page",
|
||||||
|
}
|
||||||
|
elif name in {"GitbookLoader"}:
|
||||||
|
signature["template"]["web_page"] = {
|
||||||
|
"type": "str",
|
||||||
|
"required": True,
|
||||||
|
"show": True,
|
||||||
|
"name": "web_page",
|
||||||
|
"value": "",
|
||||||
|
"display_name": "Web Page",
|
||||||
|
}
|
||||||
|
elif name in {"ReadTheDocsLoader"}:
|
||||||
|
signature["template"]["path"] = {
|
||||||
|
"type": "str",
|
||||||
|
"required": True,
|
||||||
|
"show": True,
|
||||||
|
"name": "path",
|
||||||
|
"value": "",
|
||||||
|
"display_name": "Web Page",
|
||||||
}
|
}
|
||||||
|
|
||||||
return signature
|
return signature
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
raise ValueError(f"Documment Loader {name} not found") from exc
|
raise ValueError(f"Document Loader {name} not found") from exc
|
||||||
|
|
||||||
def to_list(self) -> List[str]:
|
def to_list(self) -> List[str]:
|
||||||
return [
|
return [
|
||||||
|
|
|
||||||
|
|
@ -1,22 +0,0 @@
|
||||||
"""Load text files."""
|
|
||||||
from typing import List
|
|
||||||
|
|
||||||
from langchain.docstore.document import Document
|
|
||||||
from langchain.document_loaders.base import BaseLoader
|
|
||||||
|
|
||||||
|
|
||||||
class TextLoader(BaseLoader):
|
|
||||||
"""Load Text files."""
|
|
||||||
|
|
||||||
def __init__(self, file: str):
|
|
||||||
"""Initialize with file path."""
|
|
||||||
self.file = file
|
|
||||||
|
|
||||||
def load(self) -> List[Document]:
|
|
||||||
"""Load from file path."""
|
|
||||||
return [Document(page_content=self.file, metadata={"source": "loaded"})]
|
|
||||||
|
|
||||||
|
|
||||||
CUSTOM_DOCUMENTLOADERS = {
|
|
||||||
"TextLoader": TextLoader,
|
|
||||||
}
|
|
||||||
|
|
@ -10,7 +10,6 @@ from langchain.chat_models.base import BaseChatModel
|
||||||
from langchain.llms.base import BaseLLM
|
from langchain.llms.base import BaseLLM
|
||||||
from langchain.tools import BaseTool
|
from langchain.tools import BaseTool
|
||||||
|
|
||||||
from langflow.interface.documentLoaders.custom import CUSTOM_DOCUMENTLOADERS
|
|
||||||
from langflow.interface.tools.util import get_tool_by_name
|
from langflow.interface.tools.util import get_tool_by_name
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -132,8 +131,6 @@ def import_vectorstore(vectorstore: str) -> Any:
|
||||||
|
|
||||||
def import_documentloader(documentloader: str) -> Any:
|
def import_documentloader(documentloader: str) -> Any:
|
||||||
"""Import documentloader from documentloader name"""
|
"""Import documentloader from documentloader name"""
|
||||||
if documentloader in CUSTOM_DOCUMENTLOADERS:
|
|
||||||
return CUSTOM_DOCUMENTLOADERS[documentloader]
|
|
||||||
|
|
||||||
return import_class(f"langchain.document_loaders.{documentloader}")
|
return import_class(f"langchain.document_loaders.{documentloader}")
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -22,6 +22,7 @@ from langflow.interface.agents.custom import CUSTOM_AGENTS
|
||||||
from langflow.interface.importing.utils import import_by_type
|
from langflow.interface.importing.utils import import_by_type
|
||||||
from langflow.interface.toolkits.base import toolkits_creator
|
from langflow.interface.toolkits.base import toolkits_creator
|
||||||
from langflow.interface.types import get_type_list
|
from langflow.interface.types import get_type_list
|
||||||
|
from langflow.interface.utils import load_file_into_dict
|
||||||
from langflow.utils import util, validate
|
from langflow.utils import util, validate
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -36,21 +37,25 @@ def instantiate_class(node_type: str, base_type: str, params: Dict) -> Any:
|
||||||
if base_type == "agents":
|
if base_type == "agents":
|
||||||
# We need to initialize it differently
|
# We need to initialize it differently
|
||||||
return load_agent_executor(class_object, params)
|
return load_agent_executor(class_object, params)
|
||||||
elif node_type == "ZeroShotPrompt":
|
elif base_type == "prompts":
|
||||||
if "tools" not in params:
|
if node_type == "ZeroShotPrompt":
|
||||||
params["tools"] = []
|
if "tools" not in params:
|
||||||
return ZeroShotAgent.create_prompt(**params)
|
params["tools"] = []
|
||||||
|
return ZeroShotAgent.create_prompt(**params)
|
||||||
elif node_type == "PythonFunction":
|
elif base_type == "tools":
|
||||||
# If the node_type is "PythonFunction"
|
if node_type == "JsonSpec":
|
||||||
# we need to get the function from the params
|
params["dict_"] = load_file_into_dict(params.pop("path"))
|
||||||
# which will be a str containing a python function
|
return class_object(**params)
|
||||||
# and then we need to compile it and return the function
|
elif node_type == "PythonFunction":
|
||||||
# as the instance
|
# If the node_type is "PythonFunction"
|
||||||
function_string = params["code"]
|
# we need to get the function from the params
|
||||||
if isinstance(function_string, str):
|
# which will be a str containing a python function
|
||||||
return validate.eval_function(function_string)
|
# and then we need to compile it and return the function
|
||||||
raise ValueError("Function should be a string")
|
# as the instance
|
||||||
|
function_string = params["code"]
|
||||||
|
if isinstance(function_string, str):
|
||||||
|
return validate.eval_function(function_string)
|
||||||
|
raise ValueError("Function should be a string")
|
||||||
elif base_type == "toolkits":
|
elif base_type == "toolkits":
|
||||||
loaded_toolkit = class_object(**params)
|
loaded_toolkit = class_object(**params)
|
||||||
# Check if node_type has a loader
|
# Check if node_type has a loader
|
||||||
|
|
@ -68,8 +73,8 @@ def instantiate_class(node_type: str, base_type: str, params: Dict) -> Any:
|
||||||
documents = params.pop("documents")
|
documents = params.pop("documents")
|
||||||
text_splitter = class_object(**params)
|
text_splitter = class_object(**params)
|
||||||
return text_splitter.split_documents(documents)
|
return text_splitter.split_documents(documents)
|
||||||
else:
|
|
||||||
return class_object(**params)
|
return class_object(**params)
|
||||||
|
|
||||||
|
|
||||||
def load_flow_from_json(path: str):
|
def load_flow_from_json(path: str):
|
||||||
|
|
|
||||||
|
|
@ -25,6 +25,15 @@ class TextSplitterCreator(LangChainTypeCreator):
|
||||||
"name": "documents",
|
"name": "documents",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
signature["template"]["separator"] = {
|
||||||
|
"type": "str",
|
||||||
|
"required": True,
|
||||||
|
"show": True,
|
||||||
|
"value": ".",
|
||||||
|
"name": "separator",
|
||||||
|
"display_name": "Separator",
|
||||||
|
}
|
||||||
|
|
||||||
return signature
|
return signature
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
raise ValueError(f"Text Splitter {name} not found") from exc
|
raise ValueError(f"Text Splitter {name} not found") from exc
|
||||||
|
|
|
||||||
|
|
@ -47,12 +47,14 @@ TOOL_INPUTS = {
|
||||||
value="",
|
value="",
|
||||||
multiline=True,
|
multiline=True,
|
||||||
),
|
),
|
||||||
"dict_": TemplateField(
|
"path": TemplateField(
|
||||||
field_type="file",
|
field_type="file",
|
||||||
required=True,
|
required=True,
|
||||||
is_list=False,
|
is_list=False,
|
||||||
show=True,
|
show=True,
|
||||||
value="",
|
value="",
|
||||||
|
suffixes=[".json", ".yaml", ".yml"],
|
||||||
|
fileTypes=["json", "yaml", "yml"],
|
||||||
),
|
),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -114,6 +116,8 @@ class ToolCreator(LangChainTypeCreator):
|
||||||
return node
|
return node
|
||||||
elif tool_type in FILE_TOOLS:
|
elif tool_type in FILE_TOOLS:
|
||||||
params = all_tools[name]["params"] # type: ignore
|
params = all_tools[name]["params"] # type: ignore
|
||||||
|
if tool_type == "JsonSpec":
|
||||||
|
params["path"] = params.pop("dict_") # type: ignore
|
||||||
base_classes += [name]
|
base_classes += [name]
|
||||||
else:
|
else:
|
||||||
params = []
|
params = []
|
||||||
|
|
|
||||||
22
src/backend/langflow/interface/utils.py
Normal file
22
src/backend/langflow/interface/utils.py
Normal file
|
|
@ -0,0 +1,22 @@
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
|
||||||
|
import yaml
|
||||||
|
|
||||||
|
|
||||||
|
def load_file_into_dict(file_path: str) -> dict:
|
||||||
|
if not os.path.exists(file_path):
|
||||||
|
raise FileNotFoundError(f"File not found: {file_path}")
|
||||||
|
|
||||||
|
file_extension = os.path.splitext(file_path)[1].lower()
|
||||||
|
|
||||||
|
if file_extension == ".json":
|
||||||
|
with open(file_path, "r") as json_file:
|
||||||
|
data = json.load(json_file)
|
||||||
|
elif file_extension in [".yaml", ".yml"]:
|
||||||
|
with open(file_path, "r") as yaml_file:
|
||||||
|
data = yaml.safe_load(yaml_file)
|
||||||
|
else:
|
||||||
|
raise ValueError("Unsupported file type. Please provide a JSON or YAML file.")
|
||||||
|
|
||||||
|
return data
|
||||||
Loading…
Add table
Add a link
Reference in a new issue