From 47753d37d39daf7e3b1c55821827d12c33312370 Mon Sep 17 00:00:00 2001 From: Cristhian Zanforlin Lousa Date: Mon, 10 Mar 2025 09:28:51 -0300 Subject: [PATCH] feat: Add JSON field extraction and enhanced URL validation (#6051) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * URL component improvement - JSON URL * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * ♻️ (url.py): refactor URLComponent class to simplify data_dict creation by using dictionary unpacking instead of manual key-value pairs * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * 📝 (url.py): improve formatting of info string for DropdownInput in URLComponent class ♻️ (url.py): refactor ensure_url method to simplify logic and improve readability 🐛 (url.py): fix error handling in URLComponent class for invalid JSON content * [autofix.ci] apply automated fixes * ✨ (url.py): Add BoolInput and StrInput to support new features in URLComponent 📝 (url.py): Update description in URLComponent to provide more detailed information about its functionality ♻️ (url.py): Refactor update_build_config method in URLComponent to dynamically update fields based on selected format 🐛 (url.py): Fix ensure_url method in URLComponent to ensure valid URLs are provided and handle exceptions properly 🐛 (url.py): Fix fetch_content method in URLComponent to handle cases where no valid URLs are provided and improve error handling 🐛 (url.py): Fix fetch_content_text method in URLComponent to correctly format and clean text output based on selected format and settings 🐛 (url.py): Fix as_dataframe method in URLComponent to return fetched content as a DataFrame object * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * ♻️ (url.py): remove unnecessary comments and improve code readability by removing redundant comments and adjusting code structure. * [autofix.ci] apply automated fixes * 📝 (url.py): improve readability by splitting long description and info strings into multiple lines 🐛 (url.py): handle cases where invalid URLs or JSON URLs are provided, and provide informative error messages 🐛 (url.py): handle cases where no valid URLs are provided and raise an error with a clear message * 🔧 (Blog Writer.json, Custom Component Maker.json, Graph Vector Store RAG.json): resolve merge conflicts in JSON files related to the 'format' field options to ensure consistency across starter projects. * [autofix.ci] apply automated fixes * 🐛 (url.py): fix validation of JSON content from URLs to ensure correct handling of JSON data ✨ (url.py): introduce async validation of JSON content from URLs using aiohttp to improve performance and reliability * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * merge fix * ✅ (test_audio_file.wav): update test_audio_file.wav to fix binary file differences in the test asset * 🐛 (test_url_component.py): update error message format to improve clarity and consistency * update templates * 🐛 (test_database.py): fix error handling in test_read_flows_components_only_paginated to properly catch and log exceptions during test execution * 📝 (backend): Add noqa comments to files to ignore specific linting rule A005 🔧 (test_database.py): Remove duplicate import statement for sqlalchemy ♻️ (test_database.py): Refactor test_read_flows_components_only_paginated function for better readability and maintainability * 🐛 (test_database.py): fix test_read_flows_components_only_paginated to handle exceptions and provide more context in case of failure * 📝 (test_database.py): remove unnecessary comment to improve code readability and maintainability * 📝 (backend): Remove unnecessary noqa comments from __init__.py files 🔧 (test_database.py): Refactor test_read_flows_components_only_paginated function for better readability and maintainability * [autofix.ci] apply automated fixes --------- Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com> --- .../base/langflow/components/data/url.py | 149 +++++++--- .../starter_projects/Blog Writer.json | 46 ++- .../Custom Component Maker.json | 138 ++++++++- .../Graph Vector Store RAG.json | 46 ++- .../starter_projects/Simple Agent.json | 225 +++++++++----- .../Travel Planning Agents.json | 279 +++++++++++------- .../components/data/test_url_component.py | 2 +- src/backend/tests/unit/test_database.py | 3 + src/frontend/tests/assets/test_audio_file.wav | Bin 3249863 -> 3249862 bytes .../extended/features/loop-component.spec.ts | 2 +- 10 files changed, 638 insertions(+), 252 deletions(-) diff --git a/src/backend/base/langflow/components/data/url.py b/src/backend/base/langflow/components/data/url.py index 1adf6e086..460a99148 100644 --- a/src/backend/base/langflow/components/data/url.py +++ b/src/backend/base/langflow/components/data/url.py @@ -1,10 +1,12 @@ +import asyncio +import json import re +import aiohttp from langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader from langflow.custom import Component -from langflow.helpers.data import data_to_text -from langflow.io import DropdownInput, MessageTextInput, Output +from langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput from langflow.schema import Data from langflow.schema.dataframe import DataFrame from langflow.schema.message import Message @@ -12,7 +14,10 @@ from langflow.schema.message import Message class URLComponent(Component): display_name = "URL" - description = "Load and retrive data from specified URLs." + description = ( + "Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, " + "or JSON, with options for cleaning and separating multiple outputs." + ) icon = "layout-template" name = "URL" @@ -28,69 +33,143 @@ class URLComponent(Component): DropdownInput( name="format", display_name="Output Format", - info="Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", - options=["Text", "Raw HTML"], + info=( + "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML " + "content, or 'JSON' to extract JSON from the HTML." + ), + options=["Text", "Raw HTML", "JSON"], value="Text", + real_time_refresh=True, + ), + StrInput( + name="separator", + display_name="Separator", + value="\n\n", + show=True, + info=( + "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. " + "Default for Raw HTML is '\\n\\n'." + ), + ), + BoolInput( + name="clean_extra_whitespace", + display_name="Clean Extra Whitespace", + value=True, + show=True, + info="Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", ), ] outputs = [ Output(display_name="Data", name="data", method="fetch_content"), - Output(display_name="Message", name="text", method="fetch_content_text"), + Output(display_name="Text", name="text", method="fetch_content_text"), Output(display_name="DataFrame", name="dataframe", method="as_dataframe"), ] + async def validate_json_content(self, url: str) -> bool: + """Validates if the URL content is actually JSON.""" + try: + async with aiohttp.ClientSession() as session, session.get(url) as response: + http_ok = 200 + if response.status != http_ok: + return False + + content = await response.text() + try: + json.loads(content) + except json.JSONDecodeError: + return False + else: + return True + except (aiohttp.ClientError, asyncio.TimeoutError): + # Log specific error for debugging if needed + return False + + def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict: + """Dynamically update fields based on selected format.""" + if field_name == "format": + is_text_mode = field_value == "Text" + is_json_mode = field_value == "JSON" + build_config["separator"]["value"] = "\n\n" if is_text_mode else "\n\n" + build_config["clean_extra_whitespace"]["show"] = is_text_mode + build_config["separator"]["show"] = not is_json_mode + return build_config + def ensure_url(self, string: str) -> str: - """Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'. - - Raises an error if the string is not a valid URL. - - Parameters: - string (str): The string to be checked and possibly modified. - - Returns: - str: The modified string that is ensured to be a URL. - - Raises: - ValueError: If the string is not a valid URL. - """ + """Ensures the given string is a valid URL.""" if not string.startswith(("http://", "https://")): string = "http://" + string - # Basic URL validation regex url_regex = re.compile( - r"^(https?:\/\/)?" # optional protocol - r"(www\.)?" # optional www - r"([a-zA-Z0-9.-]+)" # domain - r"(\.[a-zA-Z]{2,})?" # top-level domain - r"(:\d+)?" # optional port - r"(\/[^\s]*)?$", # optional path + r"^(https?:\/\/)?" + r"(www\.)?" + r"([a-zA-Z0-9.-]+)" + r"(\.[a-zA-Z]{2,})?" + r"(:\d+)?" + r"(\/[^\s]*)?$", re.IGNORECASE, ) + error_msg = "Invalid URL - " + string if not url_regex.match(string): - msg = f"Invalid URL: {string}" - raise ValueError(msg) + raise ValueError(error_msg) return string def fetch_content(self) -> list[Data]: - urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()] + """Fetch content based on selected format.""" + urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()}) + + no_urls_msg = "No valid URLs provided." + if not urls: + raise ValueError(no_urls_msg) + + # If JSON format is selected, validate JSON content first + if self.format == "JSON": + for url in urls: + is_json = asyncio.run(self.validate_json_content(url)) + if not is_json: + error_msg = "Invalid JSON content from URL - " + url + raise ValueError(error_msg) + if self.format == "Raw HTML": loader = AsyncHtmlLoader(web_path=urls, encoding="utf-8") else: loader = WebBaseLoader(web_paths=urls, encoding="utf-8") + docs = loader.load() - data = [Data(text=doc.page_content, **doc.metadata) for doc in docs] - self.status = data - return data + + if self.format == "JSON": + data = [] + for doc in docs: + try: + json_content = json.loads(doc.page_content) + data_dict = {"text": json.dumps(json_content, indent=2), **json_content, **doc.metadata} + data.append(Data(**data_dict)) + except json.JSONDecodeError as err: + source = doc.metadata.get("source", "unknown URL") + error_msg = "Invalid JSON content from " + source + raise ValueError(error_msg) from err + return data + + return [Data(text=doc.page_content, **doc.metadata) for doc in docs] def fetch_content_text(self) -> Message: + """Fetch content and return as formatted text.""" data = self.fetch_content() - result_string = data_to_text("{text}", data) - self.status = result_string - return Message(text=result_string) + if self.format == "JSON": + text_list = [item.text for item in data] + result = "\n".join(text_list) + else: + text_list = [item.text for item in data] + if self.format == "Text" and self.clean_extra_whitespace: + text_list = [re.sub(r"\n{3,}", "\n\n", text) for text in text_list] + result = self.separator.join(text_list) + + self.status = result + return Message(text=result) def as_dataframe(self) -> DataFrame: + """Return fetched content as a DataFrame.""" return DataFrame(self.fetch_content()) diff --git a/src/backend/base/langflow/initial_setup/starter_projects/Blog Writer.json b/src/backend/base/langflow/initial_setup/starter_projects/Blog Writer.json index e3d20019e..70b28bfee 100644 --- a/src/backend/base/langflow/initial_setup/starter_projects/Blog Writer.json +++ b/src/backend/base/langflow/initial_setup/starter_projects/Blog Writer.json @@ -1399,7 +1399,7 @@ { "allows_loop": false, "cache": true, - "display_name": "Message", + "display_name": "Text", "method": "fetch_content_text", "name": "text", "selected": "Message", @@ -1427,6 +1427,24 @@ "score": 2.220446049250313e-16, "template": { "_type": "Component", + "clean_extra_whitespace": { + "_input_type": "BoolInput", + "advanced": false, + "display_name": "Clean Extra Whitespace", + "dynamic": false, + "info": "Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", + "list": false, + "list_add_label": "Add More", + "name": "clean_extra_whitespace", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "bool", + "value": true + }, "code": { "advanced": true, "dynamic": true, @@ -1443,7 +1461,7 @@ "show": true, "title_case": false, "type": "code", - "value": "import re\n\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.helpers.data import data_to_text\nfrom langflow.io import DropdownInput, MessageTextInput, Output\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = \"Load and retrive data from specified URLs.\"\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=\"Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.\",\n options=[\"Text\", \"Raw HTML\"],\n value=\"Text\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Message\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.\n\n Raises an error if the string is not a valid URL.\n\n Parameters:\n string (str): The string to be checked and possibly modified.\n\n Returns:\n str: The modified string that is ensured to be a URL.\n\n Raises:\n ValueError: If the string is not a valid URL.\n \"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n # Basic URL validation regex\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\" # optional protocol\n r\"(www\\.)?\" # optional www\n r\"([a-zA-Z0-9.-]+)\" # domain\n r\"(\\.[a-zA-Z]{2,})?\" # top-level domain\n r\"(:\\d+)?\" # optional port\n r\"(\\/[^\\s]*)?$\", # optional path\n re.IGNORECASE,\n )\n\n if not url_regex.match(string):\n msg = f\"Invalid URL: {string}\"\n raise ValueError(msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n docs = loader.load()\n data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n self.status = data\n return data\n\n def fetch_content_text(self) -> Message:\n data = self.fetch_content()\n\n result_string = data_to_text(\"{text}\", data)\n self.status = result_string\n return Message(text=result_string)\n\n def as_dataframe(self) -> DataFrame:\n return DataFrame(self.fetch_content())\n" + "value": "import asyncio\nimport json\nimport re\n\nimport aiohttp\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = (\n \"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, \"\n \"or JSON, with options for cleaning and separating multiple outputs.\"\n )\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=(\n \"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML \"\n \"content, or 'JSON' to extract JSON from the HTML.\"\n ),\n options=[\"Text\", \"Raw HTML\", \"JSON\"],\n value=\"Text\",\n real_time_refresh=True,\n ),\n StrInput(\n name=\"separator\",\n display_name=\"Separator\",\n value=\"\\n\\n\",\n show=True,\n info=(\n \"Specify the separator to use between multiple outputs. Default for Text is '\\\\n\\\\n'. \"\n \"Default for Raw HTML is '\\\\n\\\\n'.\"\n ),\n ),\n BoolInput(\n name=\"clean_extra_whitespace\",\n display_name=\"Clean Extra Whitespace\",\n value=True,\n show=True,\n info=\"Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Text\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n async def validate_json_content(self, url: str) -> bool:\n \"\"\"Validates if the URL content is actually JSON.\"\"\"\n try:\n async with aiohttp.ClientSession() as session, session.get(url) as response:\n http_ok = 200\n if response.status != http_ok:\n return False\n\n content = await response.text()\n try:\n json.loads(content)\n except json.JSONDecodeError:\n return False\n else:\n return True\n except (aiohttp.ClientError, asyncio.TimeoutError):\n # Log specific error for debugging if needed\n return False\n\n def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:\n \"\"\"Dynamically update fields based on selected format.\"\"\"\n if field_name == \"format\":\n is_text_mode = field_value == \"Text\"\n is_json_mode = field_value == \"JSON\"\n build_config[\"separator\"][\"value\"] = \"\\n\\n\" if is_text_mode else \"\\n\\n\"\n build_config[\"clean_extra_whitespace\"][\"show\"] = is_text_mode\n build_config[\"separator\"][\"show\"] = not is_json_mode\n return build_config\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a valid URL.\"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\"\n r\"(www\\.)?\"\n r\"([a-zA-Z0-9.-]+)\"\n r\"(\\.[a-zA-Z]{2,})?\"\n r\"(:\\d+)?\"\n r\"(\\/[^\\s]*)?$\",\n re.IGNORECASE,\n )\n\n error_msg = \"Invalid URL - \" + string\n if not url_regex.match(string):\n raise ValueError(error_msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n \"\"\"Fetch content based on selected format.\"\"\"\n urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})\n\n no_urls_msg = \"No valid URLs provided.\"\n if not urls:\n raise ValueError(no_urls_msg)\n\n # If JSON format is selected, validate JSON content first\n if self.format == \"JSON\":\n for url in urls:\n is_json = asyncio.run(self.validate_json_content(url))\n if not is_json:\n error_msg = \"Invalid JSON content from URL - \" + url\n raise ValueError(error_msg)\n\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n\n docs = loader.load()\n\n if self.format == \"JSON\":\n data = []\n for doc in docs:\n try:\n json_content = json.loads(doc.page_content)\n data_dict = {\"text\": json.dumps(json_content, indent=2), **json_content, **doc.metadata}\n data.append(Data(**data_dict))\n except json.JSONDecodeError as err:\n source = doc.metadata.get(\"source\", \"unknown URL\")\n error_msg = \"Invalid JSON content from \" + source\n raise ValueError(error_msg) from err\n return data\n\n return [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n\n def fetch_content_text(self) -> Message:\n \"\"\"Fetch content and return as formatted text.\"\"\"\n data = self.fetch_content()\n\n if self.format == \"JSON\":\n text_list = [item.text for item in data]\n result = \"\\n\".join(text_list)\n else:\n text_list = [item.text for item in data]\n if self.format == \"Text\" and self.clean_extra_whitespace:\n text_list = [re.sub(r\"\\n{3,}\", \"\\n\\n\", text) for text in text_list]\n result = self.separator.join(text_list)\n\n self.status = result\n return Message(text=result)\n\n def as_dataframe(self) -> DataFrame:\n \"\"\"Return fetched content as a DataFrame.\"\"\"\n return DataFrame(self.fetch_content())\n" }, "format": { "_input_type": "DropdownInput", @@ -1452,11 +1470,12 @@ "dialog_inputs": {}, "display_name": "Output Format", "dynamic": false, - "info": "Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", + "info": "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML content, or 'JSON' to extract JSON from the HTML.", "name": "format", "options": [ "Text", - "Raw HTML" + "Raw HTML", + "JSON" ], "options_metadata": [], "placeholder": "", @@ -1468,6 +1487,25 @@ "type": "str", "value": "Text" }, + "separator": { + "_input_type": "StrInput", + "advanced": false, + "display_name": "Separator", + "dynamic": false, + "info": "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. Default for Raw HTML is '\\n\\n'.", + "list": false, + "list_add_label": "Add More", + "load_from_db": false, + "name": "separator", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "str", + "value": "\n\n" + }, "urls": { "_input_type": "MessageTextInput", "advanced": false, diff --git a/src/backend/base/langflow/initial_setup/starter_projects/Custom Component Maker.json b/src/backend/base/langflow/initial_setup/starter_projects/Custom Component Maker.json index d6202ced7..82733dcea 100644 --- a/src/backend/base/langflow/initial_setup/starter_projects/Custom Component Maker.json +++ b/src/backend/base/langflow/initial_setup/starter_projects/Custom Component Maker.json @@ -1409,7 +1409,7 @@ { "allows_loop": false, "cache": true, - "display_name": "Message", + "display_name": "Text", "method": "fetch_content_text", "name": "text", "selected": "Message", @@ -1436,6 +1436,24 @@ "pinned": false, "template": { "_type": "Component", + "clean_extra_whitespace": { + "_input_type": "BoolInput", + "advanced": false, + "display_name": "Clean Extra Whitespace", + "dynamic": false, + "info": "Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", + "list": false, + "list_add_label": "Add More", + "name": "clean_extra_whitespace", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "bool", + "value": true + }, "code": { "advanced": true, "dynamic": true, @@ -1452,7 +1470,7 @@ "show": true, "title_case": false, "type": "code", - "value": "import re\n\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.helpers.data import data_to_text\nfrom langflow.io import DropdownInput, MessageTextInput, Output\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = \"Load and retrive data from specified URLs.\"\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=\"Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.\",\n options=[\"Text\", \"Raw HTML\"],\n value=\"Text\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Message\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.\n\n Raises an error if the string is not a valid URL.\n\n Parameters:\n string (str): The string to be checked and possibly modified.\n\n Returns:\n str: The modified string that is ensured to be a URL.\n\n Raises:\n ValueError: If the string is not a valid URL.\n \"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n # Basic URL validation regex\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\" # optional protocol\n r\"(www\\.)?\" # optional www\n r\"([a-zA-Z0-9.-]+)\" # domain\n r\"(\\.[a-zA-Z]{2,})?\" # top-level domain\n r\"(:\\d+)?\" # optional port\n r\"(\\/[^\\s]*)?$\", # optional path\n re.IGNORECASE,\n )\n\n if not url_regex.match(string):\n msg = f\"Invalid URL: {string}\"\n raise ValueError(msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n docs = loader.load()\n data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n self.status = data\n return data\n\n def fetch_content_text(self) -> Message:\n data = self.fetch_content()\n\n result_string = data_to_text(\"{text}\", data)\n self.status = result_string\n return Message(text=result_string)\n\n def as_dataframe(self) -> DataFrame:\n return DataFrame(self.fetch_content())\n" + "value": "import asyncio\nimport json\nimport re\n\nimport aiohttp\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = (\n \"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, \"\n \"or JSON, with options for cleaning and separating multiple outputs.\"\n )\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=(\n \"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML \"\n \"content, or 'JSON' to extract JSON from the HTML.\"\n ),\n options=[\"Text\", \"Raw HTML\", \"JSON\"],\n value=\"Text\",\n real_time_refresh=True,\n ),\n StrInput(\n name=\"separator\",\n display_name=\"Separator\",\n value=\"\\n\\n\",\n show=True,\n info=(\n \"Specify the separator to use between multiple outputs. Default for Text is '\\\\n\\\\n'. \"\n \"Default for Raw HTML is '\\\\n\\\\n'.\"\n ),\n ),\n BoolInput(\n name=\"clean_extra_whitespace\",\n display_name=\"Clean Extra Whitespace\",\n value=True,\n show=True,\n info=\"Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Text\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n async def validate_json_content(self, url: str) -> bool:\n \"\"\"Validates if the URL content is actually JSON.\"\"\"\n try:\n async with aiohttp.ClientSession() as session, session.get(url) as response:\n http_ok = 200\n if response.status != http_ok:\n return False\n\n content = await response.text()\n try:\n json.loads(content)\n except json.JSONDecodeError:\n return False\n else:\n return True\n except (aiohttp.ClientError, asyncio.TimeoutError):\n # Log specific error for debugging if needed\n return False\n\n def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:\n \"\"\"Dynamically update fields based on selected format.\"\"\"\n if field_name == \"format\":\n is_text_mode = field_value == \"Text\"\n is_json_mode = field_value == \"JSON\"\n build_config[\"separator\"][\"value\"] = \"\\n\\n\" if is_text_mode else \"\\n\\n\"\n build_config[\"clean_extra_whitespace\"][\"show\"] = is_text_mode\n build_config[\"separator\"][\"show\"] = not is_json_mode\n return build_config\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a valid URL.\"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\"\n r\"(www\\.)?\"\n r\"([a-zA-Z0-9.-]+)\"\n r\"(\\.[a-zA-Z]{2,})?\"\n r\"(:\\d+)?\"\n r\"(\\/[^\\s]*)?$\",\n re.IGNORECASE,\n )\n\n error_msg = \"Invalid URL - \" + string\n if not url_regex.match(string):\n raise ValueError(error_msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n \"\"\"Fetch content based on selected format.\"\"\"\n urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})\n\n no_urls_msg = \"No valid URLs provided.\"\n if not urls:\n raise ValueError(no_urls_msg)\n\n # If JSON format is selected, validate JSON content first\n if self.format == \"JSON\":\n for url in urls:\n is_json = asyncio.run(self.validate_json_content(url))\n if not is_json:\n error_msg = \"Invalid JSON content from URL - \" + url\n raise ValueError(error_msg)\n\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n\n docs = loader.load()\n\n if self.format == \"JSON\":\n data = []\n for doc in docs:\n try:\n json_content = json.loads(doc.page_content)\n data_dict = {\"text\": json.dumps(json_content, indent=2), **json_content, **doc.metadata}\n data.append(Data(**data_dict))\n except json.JSONDecodeError as err:\n source = doc.metadata.get(\"source\", \"unknown URL\")\n error_msg = \"Invalid JSON content from \" + source\n raise ValueError(error_msg) from err\n return data\n\n return [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n\n def fetch_content_text(self) -> Message:\n \"\"\"Fetch content and return as formatted text.\"\"\"\n data = self.fetch_content()\n\n if self.format == \"JSON\":\n text_list = [item.text for item in data]\n result = \"\\n\".join(text_list)\n else:\n text_list = [item.text for item in data]\n if self.format == \"Text\" and self.clean_extra_whitespace:\n text_list = [re.sub(r\"\\n{3,}\", \"\\n\\n\", text) for text in text_list]\n result = self.separator.join(text_list)\n\n self.status = result\n return Message(text=result)\n\n def as_dataframe(self) -> DataFrame:\n \"\"\"Return fetched content as a DataFrame.\"\"\"\n return DataFrame(self.fetch_content())\n" }, "format": { "_input_type": "DropdownInput", @@ -1460,11 +1478,12 @@ "combobox": false, "display_name": "Output Format", "dynamic": false, - "info": "Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", + "info": "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML content, or 'JSON' to extract JSON from the HTML.", "name": "format", "options": [ "Text", - "Raw HTML" + "Raw HTML", + "JSON" ], "placeholder": "", "required": false, @@ -1475,6 +1494,25 @@ "type": "str", "value": "Text" }, + "separator": { + "_input_type": "StrInput", + "advanced": false, + "display_name": "Separator", + "dynamic": false, + "info": "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. Default for Raw HTML is '\\n\\n'.", + "list": false, + "list_add_label": "Add More", + "load_from_db": false, + "name": "separator", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "str", + "value": "\n\n" + }, "urls": { "_input_type": "MessageTextInput", "advanced": false, @@ -1565,7 +1603,7 @@ { "allows_loop": false, "cache": true, - "display_name": "Message", + "display_name": "Text", "method": "fetch_content_text", "name": "text", "selected": "Message", @@ -1592,6 +1630,24 @@ "pinned": false, "template": { "_type": "Component", + "clean_extra_whitespace": { + "_input_type": "BoolInput", + "advanced": false, + "display_name": "Clean Extra Whitespace", + "dynamic": false, + "info": "Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", + "list": false, + "list_add_label": "Add More", + "name": "clean_extra_whitespace", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "bool", + "value": true + }, "code": { "advanced": true, "dynamic": true, @@ -1608,7 +1664,7 @@ "show": true, "title_case": false, "type": "code", - "value": "import re\n\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.helpers.data import data_to_text\nfrom langflow.io import DropdownInput, MessageTextInput, Output\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = \"Load and retrive data from specified URLs.\"\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=\"Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.\",\n options=[\"Text\", \"Raw HTML\"],\n value=\"Text\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Message\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.\n\n Raises an error if the string is not a valid URL.\n\n Parameters:\n string (str): The string to be checked and possibly modified.\n\n Returns:\n str: The modified string that is ensured to be a URL.\n\n Raises:\n ValueError: If the string is not a valid URL.\n \"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n # Basic URL validation regex\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\" # optional protocol\n r\"(www\\.)?\" # optional www\n r\"([a-zA-Z0-9.-]+)\" # domain\n r\"(\\.[a-zA-Z]{2,})?\" # top-level domain\n r\"(:\\d+)?\" # optional port\n r\"(\\/[^\\s]*)?$\", # optional path\n re.IGNORECASE,\n )\n\n if not url_regex.match(string):\n msg = f\"Invalid URL: {string}\"\n raise ValueError(msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n docs = loader.load()\n data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n self.status = data\n return data\n\n def fetch_content_text(self) -> Message:\n data = self.fetch_content()\n\n result_string = data_to_text(\"{text}\", data)\n self.status = result_string\n return Message(text=result_string)\n\n def as_dataframe(self) -> DataFrame:\n return DataFrame(self.fetch_content())\n" + "value": "import asyncio\nimport json\nimport re\n\nimport aiohttp\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = (\n \"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, \"\n \"or JSON, with options for cleaning and separating multiple outputs.\"\n )\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=(\n \"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML \"\n \"content, or 'JSON' to extract JSON from the HTML.\"\n ),\n options=[\"Text\", \"Raw HTML\", \"JSON\"],\n value=\"Text\",\n real_time_refresh=True,\n ),\n StrInput(\n name=\"separator\",\n display_name=\"Separator\",\n value=\"\\n\\n\",\n show=True,\n info=(\n \"Specify the separator to use between multiple outputs. Default for Text is '\\\\n\\\\n'. \"\n \"Default for Raw HTML is '\\\\n\\\\n'.\"\n ),\n ),\n BoolInput(\n name=\"clean_extra_whitespace\",\n display_name=\"Clean Extra Whitespace\",\n value=True,\n show=True,\n info=\"Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Text\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n async def validate_json_content(self, url: str) -> bool:\n \"\"\"Validates if the URL content is actually JSON.\"\"\"\n try:\n async with aiohttp.ClientSession() as session, session.get(url) as response:\n http_ok = 200\n if response.status != http_ok:\n return False\n\n content = await response.text()\n try:\n json.loads(content)\n except json.JSONDecodeError:\n return False\n else:\n return True\n except (aiohttp.ClientError, asyncio.TimeoutError):\n # Log specific error for debugging if needed\n return False\n\n def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:\n \"\"\"Dynamically update fields based on selected format.\"\"\"\n if field_name == \"format\":\n is_text_mode = field_value == \"Text\"\n is_json_mode = field_value == \"JSON\"\n build_config[\"separator\"][\"value\"] = \"\\n\\n\" if is_text_mode else \"\\n\\n\"\n build_config[\"clean_extra_whitespace\"][\"show\"] = is_text_mode\n build_config[\"separator\"][\"show\"] = not is_json_mode\n return build_config\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a valid URL.\"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\"\n r\"(www\\.)?\"\n r\"([a-zA-Z0-9.-]+)\"\n r\"(\\.[a-zA-Z]{2,})?\"\n r\"(:\\d+)?\"\n r\"(\\/[^\\s]*)?$\",\n re.IGNORECASE,\n )\n\n error_msg = \"Invalid URL - \" + string\n if not url_regex.match(string):\n raise ValueError(error_msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n \"\"\"Fetch content based on selected format.\"\"\"\n urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})\n\n no_urls_msg = \"No valid URLs provided.\"\n if not urls:\n raise ValueError(no_urls_msg)\n\n # If JSON format is selected, validate JSON content first\n if self.format == \"JSON\":\n for url in urls:\n is_json = asyncio.run(self.validate_json_content(url))\n if not is_json:\n error_msg = \"Invalid JSON content from URL - \" + url\n raise ValueError(error_msg)\n\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n\n docs = loader.load()\n\n if self.format == \"JSON\":\n data = []\n for doc in docs:\n try:\n json_content = json.loads(doc.page_content)\n data_dict = {\"text\": json.dumps(json_content, indent=2), **json_content, **doc.metadata}\n data.append(Data(**data_dict))\n except json.JSONDecodeError as err:\n source = doc.metadata.get(\"source\", \"unknown URL\")\n error_msg = \"Invalid JSON content from \" + source\n raise ValueError(error_msg) from err\n return data\n\n return [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n\n def fetch_content_text(self) -> Message:\n \"\"\"Fetch content and return as formatted text.\"\"\"\n data = self.fetch_content()\n\n if self.format == \"JSON\":\n text_list = [item.text for item in data]\n result = \"\\n\".join(text_list)\n else:\n text_list = [item.text for item in data]\n if self.format == \"Text\" and self.clean_extra_whitespace:\n text_list = [re.sub(r\"\\n{3,}\", \"\\n\\n\", text) for text in text_list]\n result = self.separator.join(text_list)\n\n self.status = result\n return Message(text=result)\n\n def as_dataframe(self) -> DataFrame:\n \"\"\"Return fetched content as a DataFrame.\"\"\"\n return DataFrame(self.fetch_content())\n" }, "format": { "_input_type": "DropdownInput", @@ -1616,11 +1672,12 @@ "combobox": false, "display_name": "Output Format", "dynamic": false, - "info": "Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", + "info": "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML content, or 'JSON' to extract JSON from the HTML.", "name": "format", "options": [ "Text", - "Raw HTML" + "Raw HTML", + "JSON" ], "placeholder": "", "required": false, @@ -1631,6 +1688,25 @@ "type": "str", "value": "Text" }, + "separator": { + "_input_type": "StrInput", + "advanced": false, + "display_name": "Separator", + "dynamic": false, + "info": "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. Default for Raw HTML is '\\n\\n'.", + "list": false, + "list_add_label": "Add More", + "load_from_db": false, + "name": "separator", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "str", + "value": "\n\n" + }, "urls": { "_input_type": "MessageTextInput", "advanced": false, @@ -1727,7 +1803,7 @@ { "allows_loop": false, "cache": true, - "display_name": "Message", + "display_name": "Text", "method": "fetch_content_text", "name": "text", "selected": "Message", @@ -1754,6 +1830,24 @@ "pinned": false, "template": { "_type": "Component", + "clean_extra_whitespace": { + "_input_type": "BoolInput", + "advanced": false, + "display_name": "Clean Extra Whitespace", + "dynamic": false, + "info": "Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", + "list": false, + "list_add_label": "Add More", + "name": "clean_extra_whitespace", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "bool", + "value": true + }, "code": { "advanced": true, "dynamic": true, @@ -1770,7 +1864,7 @@ "show": true, "title_case": false, "type": "code", - "value": "import re\n\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.helpers.data import data_to_text\nfrom langflow.io import DropdownInput, MessageTextInput, Output\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = \"Load and retrive data from specified URLs.\"\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=\"Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.\",\n options=[\"Text\", \"Raw HTML\"],\n value=\"Text\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Message\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.\n\n Raises an error if the string is not a valid URL.\n\n Parameters:\n string (str): The string to be checked and possibly modified.\n\n Returns:\n str: The modified string that is ensured to be a URL.\n\n Raises:\n ValueError: If the string is not a valid URL.\n \"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n # Basic URL validation regex\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\" # optional protocol\n r\"(www\\.)?\" # optional www\n r\"([a-zA-Z0-9.-]+)\" # domain\n r\"(\\.[a-zA-Z]{2,})?\" # top-level domain\n r\"(:\\d+)?\" # optional port\n r\"(\\/[^\\s]*)?$\", # optional path\n re.IGNORECASE,\n )\n\n if not url_regex.match(string):\n msg = f\"Invalid URL: {string}\"\n raise ValueError(msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n docs = loader.load()\n data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n self.status = data\n return data\n\n def fetch_content_text(self) -> Message:\n data = self.fetch_content()\n\n result_string = data_to_text(\"{text}\", data)\n self.status = result_string\n return Message(text=result_string)\n\n def as_dataframe(self) -> DataFrame:\n return DataFrame(self.fetch_content())\n" + "value": "import asyncio\nimport json\nimport re\n\nimport aiohttp\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = (\n \"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, \"\n \"or JSON, with options for cleaning and separating multiple outputs.\"\n )\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=(\n \"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML \"\n \"content, or 'JSON' to extract JSON from the HTML.\"\n ),\n options=[\"Text\", \"Raw HTML\", \"JSON\"],\n value=\"Text\",\n real_time_refresh=True,\n ),\n StrInput(\n name=\"separator\",\n display_name=\"Separator\",\n value=\"\\n\\n\",\n show=True,\n info=(\n \"Specify the separator to use between multiple outputs. Default for Text is '\\\\n\\\\n'. \"\n \"Default for Raw HTML is '\\\\n\\\\n'.\"\n ),\n ),\n BoolInput(\n name=\"clean_extra_whitespace\",\n display_name=\"Clean Extra Whitespace\",\n value=True,\n show=True,\n info=\"Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Text\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n async def validate_json_content(self, url: str) -> bool:\n \"\"\"Validates if the URL content is actually JSON.\"\"\"\n try:\n async with aiohttp.ClientSession() as session, session.get(url) as response:\n http_ok = 200\n if response.status != http_ok:\n return False\n\n content = await response.text()\n try:\n json.loads(content)\n except json.JSONDecodeError:\n return False\n else:\n return True\n except (aiohttp.ClientError, asyncio.TimeoutError):\n # Log specific error for debugging if needed\n return False\n\n def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:\n \"\"\"Dynamically update fields based on selected format.\"\"\"\n if field_name == \"format\":\n is_text_mode = field_value == \"Text\"\n is_json_mode = field_value == \"JSON\"\n build_config[\"separator\"][\"value\"] = \"\\n\\n\" if is_text_mode else \"\\n\\n\"\n build_config[\"clean_extra_whitespace\"][\"show\"] = is_text_mode\n build_config[\"separator\"][\"show\"] = not is_json_mode\n return build_config\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a valid URL.\"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\"\n r\"(www\\.)?\"\n r\"([a-zA-Z0-9.-]+)\"\n r\"(\\.[a-zA-Z]{2,})?\"\n r\"(:\\d+)?\"\n r\"(\\/[^\\s]*)?$\",\n re.IGNORECASE,\n )\n\n error_msg = \"Invalid URL - \" + string\n if not url_regex.match(string):\n raise ValueError(error_msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n \"\"\"Fetch content based on selected format.\"\"\"\n urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})\n\n no_urls_msg = \"No valid URLs provided.\"\n if not urls:\n raise ValueError(no_urls_msg)\n\n # If JSON format is selected, validate JSON content first\n if self.format == \"JSON\":\n for url in urls:\n is_json = asyncio.run(self.validate_json_content(url))\n if not is_json:\n error_msg = \"Invalid JSON content from URL - \" + url\n raise ValueError(error_msg)\n\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n\n docs = loader.load()\n\n if self.format == \"JSON\":\n data = []\n for doc in docs:\n try:\n json_content = json.loads(doc.page_content)\n data_dict = {\"text\": json.dumps(json_content, indent=2), **json_content, **doc.metadata}\n data.append(Data(**data_dict))\n except json.JSONDecodeError as err:\n source = doc.metadata.get(\"source\", \"unknown URL\")\n error_msg = \"Invalid JSON content from \" + source\n raise ValueError(error_msg) from err\n return data\n\n return [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n\n def fetch_content_text(self) -> Message:\n \"\"\"Fetch content and return as formatted text.\"\"\"\n data = self.fetch_content()\n\n if self.format == \"JSON\":\n text_list = [item.text for item in data]\n result = \"\\n\".join(text_list)\n else:\n text_list = [item.text for item in data]\n if self.format == \"Text\" and self.clean_extra_whitespace:\n text_list = [re.sub(r\"\\n{3,}\", \"\\n\\n\", text) for text in text_list]\n result = self.separator.join(text_list)\n\n self.status = result\n return Message(text=result)\n\n def as_dataframe(self) -> DataFrame:\n \"\"\"Return fetched content as a DataFrame.\"\"\"\n return DataFrame(self.fetch_content())\n" }, "format": { "_input_type": "DropdownInput", @@ -1778,11 +1872,12 @@ "combobox": false, "display_name": "Output Format", "dynamic": false, - "info": "Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", + "info": "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML content, or 'JSON' to extract JSON from the HTML.", "name": "format", "options": [ "Text", - "Raw HTML" + "Raw HTML", + "JSON" ], "placeholder": "", "required": false, @@ -1793,6 +1888,25 @@ "type": "str", "value": "Text" }, + "separator": { + "_input_type": "StrInput", + "advanced": false, + "display_name": "Separator", + "dynamic": false, + "info": "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. Default for Raw HTML is '\\n\\n'.", + "list": false, + "list_add_label": "Add More", + "load_from_db": false, + "name": "separator", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "str", + "value": "\n\n" + }, "urls": { "_input_type": "MessageTextInput", "advanced": false, diff --git a/src/backend/base/langflow/initial_setup/starter_projects/Graph Vector Store RAG.json b/src/backend/base/langflow/initial_setup/starter_projects/Graph Vector Store RAG.json index e021c7e0d..96d7baaa4 100644 --- a/src/backend/base/langflow/initial_setup/starter_projects/Graph Vector Store RAG.json +++ b/src/backend/base/langflow/initial_setup/starter_projects/Graph Vector Store RAG.json @@ -2769,7 +2769,7 @@ { "allows_loop": false, "cache": true, - "display_name": "Message", + "display_name": "Text", "method": "fetch_content_text", "name": "text", "selected": "Message", @@ -2797,6 +2797,24 @@ "score": 2.220446049250313e-16, "template": { "_type": "Component", + "clean_extra_whitespace": { + "_input_type": "BoolInput", + "advanced": false, + "display_name": "Clean Extra Whitespace", + "dynamic": false, + "info": "Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", + "list": false, + "list_add_label": "Add More", + "name": "clean_extra_whitespace", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "bool", + "value": true + }, "code": { "advanced": true, "dynamic": true, @@ -2813,7 +2831,7 @@ "show": true, "title_case": false, "type": "code", - "value": "import re\n\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.helpers.data import data_to_text\nfrom langflow.io import DropdownInput, MessageTextInput, Output\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = \"Load and retrive data from specified URLs.\"\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=\"Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.\",\n options=[\"Text\", \"Raw HTML\"],\n value=\"Text\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Message\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.\n\n Raises an error if the string is not a valid URL.\n\n Parameters:\n string (str): The string to be checked and possibly modified.\n\n Returns:\n str: The modified string that is ensured to be a URL.\n\n Raises:\n ValueError: If the string is not a valid URL.\n \"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n # Basic URL validation regex\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\" # optional protocol\n r\"(www\\.)?\" # optional www\n r\"([a-zA-Z0-9.-]+)\" # domain\n r\"(\\.[a-zA-Z]{2,})?\" # top-level domain\n r\"(:\\d+)?\" # optional port\n r\"(\\/[^\\s]*)?$\", # optional path\n re.IGNORECASE,\n )\n\n if not url_regex.match(string):\n msg = f\"Invalid URL: {string}\"\n raise ValueError(msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n docs = loader.load()\n data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n self.status = data\n return data\n\n def fetch_content_text(self) -> Message:\n data = self.fetch_content()\n\n result_string = data_to_text(\"{text}\", data)\n self.status = result_string\n return Message(text=result_string)\n\n def as_dataframe(self) -> DataFrame:\n return DataFrame(self.fetch_content())\n" + "value": "import asyncio\nimport json\nimport re\n\nimport aiohttp\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = (\n \"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, \"\n \"or JSON, with options for cleaning and separating multiple outputs.\"\n )\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=(\n \"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML \"\n \"content, or 'JSON' to extract JSON from the HTML.\"\n ),\n options=[\"Text\", \"Raw HTML\", \"JSON\"],\n value=\"Text\",\n real_time_refresh=True,\n ),\n StrInput(\n name=\"separator\",\n display_name=\"Separator\",\n value=\"\\n\\n\",\n show=True,\n info=(\n \"Specify the separator to use between multiple outputs. Default for Text is '\\\\n\\\\n'. \"\n \"Default for Raw HTML is '\\\\n\\\\n'.\"\n ),\n ),\n BoolInput(\n name=\"clean_extra_whitespace\",\n display_name=\"Clean Extra Whitespace\",\n value=True,\n show=True,\n info=\"Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Text\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n async def validate_json_content(self, url: str) -> bool:\n \"\"\"Validates if the URL content is actually JSON.\"\"\"\n try:\n async with aiohttp.ClientSession() as session, session.get(url) as response:\n http_ok = 200\n if response.status != http_ok:\n return False\n\n content = await response.text()\n try:\n json.loads(content)\n except json.JSONDecodeError:\n return False\n else:\n return True\n except (aiohttp.ClientError, asyncio.TimeoutError):\n # Log specific error for debugging if needed\n return False\n\n def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:\n \"\"\"Dynamically update fields based on selected format.\"\"\"\n if field_name == \"format\":\n is_text_mode = field_value == \"Text\"\n is_json_mode = field_value == \"JSON\"\n build_config[\"separator\"][\"value\"] = \"\\n\\n\" if is_text_mode else \"\\n\\n\"\n build_config[\"clean_extra_whitespace\"][\"show\"] = is_text_mode\n build_config[\"separator\"][\"show\"] = not is_json_mode\n return build_config\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a valid URL.\"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\"\n r\"(www\\.)?\"\n r\"([a-zA-Z0-9.-]+)\"\n r\"(\\.[a-zA-Z]{2,})?\"\n r\"(:\\d+)?\"\n r\"(\\/[^\\s]*)?$\",\n re.IGNORECASE,\n )\n\n error_msg = \"Invalid URL - \" + string\n if not url_regex.match(string):\n raise ValueError(error_msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n \"\"\"Fetch content based on selected format.\"\"\"\n urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})\n\n no_urls_msg = \"No valid URLs provided.\"\n if not urls:\n raise ValueError(no_urls_msg)\n\n # If JSON format is selected, validate JSON content first\n if self.format == \"JSON\":\n for url in urls:\n is_json = asyncio.run(self.validate_json_content(url))\n if not is_json:\n error_msg = \"Invalid JSON content from URL - \" + url\n raise ValueError(error_msg)\n\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n\n docs = loader.load()\n\n if self.format == \"JSON\":\n data = []\n for doc in docs:\n try:\n json_content = json.loads(doc.page_content)\n data_dict = {\"text\": json.dumps(json_content, indent=2), **json_content, **doc.metadata}\n data.append(Data(**data_dict))\n except json.JSONDecodeError as err:\n source = doc.metadata.get(\"source\", \"unknown URL\")\n error_msg = \"Invalid JSON content from \" + source\n raise ValueError(error_msg) from err\n return data\n\n return [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n\n def fetch_content_text(self) -> Message:\n \"\"\"Fetch content and return as formatted text.\"\"\"\n data = self.fetch_content()\n\n if self.format == \"JSON\":\n text_list = [item.text for item in data]\n result = \"\\n\".join(text_list)\n else:\n text_list = [item.text for item in data]\n if self.format == \"Text\" and self.clean_extra_whitespace:\n text_list = [re.sub(r\"\\n{3,}\", \"\\n\\n\", text) for text in text_list]\n result = self.separator.join(text_list)\n\n self.status = result\n return Message(text=result)\n\n def as_dataframe(self) -> DataFrame:\n \"\"\"Return fetched content as a DataFrame.\"\"\"\n return DataFrame(self.fetch_content())\n" }, "format": { "_input_type": "DropdownInput", @@ -2821,11 +2839,12 @@ "combobox": false, "display_name": "Output Format", "dynamic": false, - "info": "Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", + "info": "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML content, or 'JSON' to extract JSON from the HTML.", "name": "format", "options": [ "Text", - "Raw HTML" + "Raw HTML", + "JSON" ], "placeholder": "", "required": false, @@ -2836,6 +2855,25 @@ "type": "str", "value": "Raw HTML" }, + "separator": { + "_input_type": "StrInput", + "advanced": false, + "display_name": "Separator", + "dynamic": false, + "info": "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. Default for Raw HTML is '\\n\\n'.", + "list": false, + "list_add_label": "Add More", + "load_from_db": false, + "name": "separator", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "str", + "value": "\n\n" + }, "urls": { "_input_type": "MessageTextInput", "advanced": false, diff --git a/src/backend/base/langflow/initial_setup/starter_projects/Simple Agent.json b/src/backend/base/langflow/initial_setup/starter_projects/Simple Agent.json index bf15a5aa8..5eeb20936 100644 --- a/src/backend/base/langflow/initial_setup/starter_projects/Simple Agent.json +++ b/src/backend/base/langflow/initial_setup/starter_projects/Simple Agent.json @@ -7,7 +7,7 @@ "data": { "sourceHandle": { "dataType": "ChatInput", - "id": "ChatInput-vSKY8", + "id": "ChatInput-5rifq", "name": "message", "output_types": [ "Message" @@ -15,19 +15,19 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "Agent-EPjLx", + "id": "Agent-eACmP", "inputTypes": [ "Message" ], "type": "str" } }, - "id": "reactflow__edge-ChatInput-vSKY8{œdataTypeœ:œChatInputœ,œidœ:œChatInput-vSKY8œ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Agent-EPjLx{œfieldNameœ:œinput_valueœ,œidœ:œAgent-EPjLxœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", + "id": "reactflow__edge-ChatInput-5rifq{œdataTypeœ:œChatInputœ,œidœ:œChatInput-5rifqœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Agent-eACmP{œfieldNameœ:œinput_valueœ,œidœ:œAgent-eACmPœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", "selected": false, - "source": "ChatInput-vSKY8", - "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-vSKY8œ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", - "target": "Agent-EPjLx", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-EPjLxœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" + "source": "ChatInput-5rifq", + "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-5rifqœ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", + "target": "Agent-eACmP", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-eACmPœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -35,7 +35,7 @@ "data": { "sourceHandle": { "dataType": "Agent", - "id": "Agent-EPjLx", + "id": "Agent-eACmP", "name": "response", "output_types": [ "Message" @@ -43,7 +43,7 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "ChatOutput-x6tZi", + "id": "ChatOutput-jYRjS", "inputTypes": [ "Data", "DataFrame", @@ -52,19 +52,20 @@ "type": "str" } }, - "id": "reactflow__edge-Agent-EPjLx{œdataTypeœ:œAgentœ,œidœ:œAgent-EPjLxœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-ChatOutput-x6tZi{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-x6tZiœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", + "id": "reactflow__edge-Agent-eACmP{œdataTypeœ:œAgentœ,œidœ:œAgent-eACmPœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-ChatOutput-jYRjS{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-jYRjSœ,œinputTypesœ:[œDataœ,œDataFrameœ,œMessageœ],œtypeœ:œstrœ}", "selected": false, - "source": "Agent-EPjLx", - "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-EPjLxœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", - "target": "ChatOutput-x6tZi", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œChatOutput-x6tZiœ, œinputTypesœ: [œDataœ, œDataFrameœ, œMessageœ], œtypeœ: œstrœ}" + "source": "Agent-eACmP", + "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-eACmPœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", + "target": "ChatOutput-jYRjS", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œChatOutput-jYRjSœ, œinputTypesœ: [œDataœ, œDataFrameœ, œMessageœ], œtypeœ: œstrœ}" }, { + "animated": false, "className": "", "data": { "sourceHandle": { "dataType": "CalculatorComponent", - "id": "CalculatorComponent-zxpIu", + "id": "CalculatorComponent-GTSkO", "name": "component_as_tool", "output_types": [ "Tool" @@ -72,25 +73,26 @@ }, "targetHandle": { "fieldName": "tools", - "id": "Agent-EPjLx", + "id": "Agent-eACmP", "inputTypes": [ "Tool" ], "type": "other" } }, - "id": "reactflow__edge-CalculatorComponent-zxpIu{œdataTypeœ:œCalculatorComponentœ,œidœ:œCalculatorComponent-zxpIuœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-EPjLx{œfieldNameœ:œtoolsœ,œidœ:œAgent-EPjLxœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", - "source": "CalculatorComponent-zxpIu", - "sourceHandle": "{œdataTypeœ: œCalculatorComponentœ, œidœ: œCalculatorComponent-zxpIuœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", - "target": "Agent-EPjLx", - "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-EPjLxœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" + "id": "reactflow__edge-CalculatorComponent-GTSkO{œdataTypeœ:œCalculatorComponentœ,œidœ:œCalculatorComponent-GTSkOœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-eACmP{œfieldNameœ:œtoolsœ,œidœ:œAgent-eACmPœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", + "source": "CalculatorComponent-GTSkO", + "sourceHandle": "{œdataTypeœ: œCalculatorComponentœ, œidœ: œCalculatorComponent-GTSkOœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", + "target": "Agent-eACmP", + "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-eACmPœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" }, { + "animated": false, "className": "", "data": { "sourceHandle": { "dataType": "URL", - "id": "URL-yM3pY", + "id": "URL-BvxUK", "name": "component_as_tool", "output_types": [ "Tool" @@ -98,18 +100,18 @@ }, "targetHandle": { "fieldName": "tools", - "id": "Agent-EPjLx", + "id": "Agent-eACmP", "inputTypes": [ "Tool" ], "type": "other" } }, - "id": "reactflow__edge-URL-yM3pY{œdataTypeœ:œURLœ,œidœ:œURL-yM3pYœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-EPjLx{œfieldNameœ:œtoolsœ,œidœ:œAgent-EPjLxœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", - "source": "URL-yM3pY", - "sourceHandle": "{œdataTypeœ: œURLœ, œidœ: œURL-yM3pYœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", - "target": "Agent-EPjLx", - "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-EPjLxœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" + "id": "reactflow__edge-URL-BvxUK{œdataTypeœ:œURLœ,œidœ:œURL-BvxUKœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-eACmP{œfieldNameœ:œtoolsœ,œidœ:œAgent-eACmPœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", + "source": "URL-BvxUK", + "sourceHandle": "{œdataTypeœ: œURLœ, œidœ: œURL-BvxUKœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", + "target": "Agent-eACmP", + "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-eACmPœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" } ], "nodes": [ @@ -117,7 +119,7 @@ "data": { "description": "Define the agent's instructions, then enter a task to complete using tools.", "display_name": "Agent", - "id": "Agent-EPjLx", + "id": "Agent-eACmP", "node": { "base_classes": [ "Message" @@ -160,7 +162,7 @@ "frozen": false, "icon": "bot", "legacy": false, - "lf_version": "1.1.1", + "lf_version": "1.2.0", "metadata": {}, "output_types": [], "outputs": [ @@ -715,10 +717,10 @@ "type": "Agent" }, "dragging": false, - "id": "Agent-EPjLx", + "id": "Agent-eACmP", "measured": { - "height": 621, - "width": 320 + "height": 698, + "width": 360 }, "position": { "x": 1652.2479633316434, @@ -729,7 +731,7 @@ }, { "data": { - "id": "ChatInput-vSKY8", + "id": "ChatInput-5rifq", "node": { "base_classes": [ "Message" @@ -755,7 +757,7 @@ "frozen": false, "icon": "MessagesSquare", "legacy": false, - "lf_version": "1.1.1", + "lf_version": "1.2.0", "metadata": {}, "output_types": [], "outputs": [ @@ -1010,10 +1012,10 @@ "type": "ChatInput" }, "dragging": false, - "id": "ChatInput-vSKY8", + "id": "ChatInput-5rifq", "measured": { - "height": 229, - "width": 320 + "height": 257, + "width": 360 }, "position": { "x": 1241.9566260691947, @@ -1026,7 +1028,7 @@ "data": { "description": "Display a chat message in the Playground.", "display_name": "Chat Output", - "id": "ChatOutput-x6tZi", + "id": "ChatOutput-jYRjS", "node": { "base_classes": [ "Message" @@ -1052,7 +1054,7 @@ "frozen": false, "icon": "MessagesSquare", "legacy": false, - "lf_version": "1.1.1", + "lf_version": "1.2.0", "metadata": {}, "output_types": [], "outputs": [ @@ -1306,23 +1308,23 @@ }, "type": "ChatOutput" }, - "id": "ChatOutput-x6tZi", + "id": "ChatOutput-jYRjS", "measured": { - "height": 229, - "width": 320 + "height": 257, + "width": 360 }, "position": { "x": 2029.726227044409, "y": 521.9624030396819 }, - "selected": false, + "selected": true, "type": "genericNode" }, { "data": { - "description": "Load and retrive data from specified URLs.", + "description": "Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "display_name": "URL", - "id": "URL-yM3pY", + "id": "URL-BvxUK", "node": { "base_classes": [ "Data", @@ -1332,22 +1334,26 @@ "beta": false, "conditional_paths": [], "custom_fields": {}, - "description": "Load and retrive data from specified URLs.", + "description": "Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "display_name": "URL", "documentation": "", "edited": false, "field_order": [ "urls", - "format" + "format", + "separator", + "clean_extra_whitespace" ], "frozen": false, "icon": "layout-template", "legacy": false, + "lf_version": "1.2.0", "metadata": {}, "minimized": false, "output_types": [], "outputs": [ { + "allows_loop": false, "cache": true, "display_name": "Toolset", "hidden": null, @@ -1355,6 +1361,7 @@ "name": "component_as_tool", "required_inputs": null, "selected": "Tool", + "tool_mode": true, "types": [ "Tool" ], @@ -1364,6 +1371,24 @@ "pinned": false, "template": { "_type": "Component", + "clean_extra_whitespace": { + "_input_type": "BoolInput", + "advanced": false, + "display_name": "Clean Extra Whitespace", + "dynamic": false, + "info": "Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", + "list": false, + "list_add_label": "Add More", + "name": "clean_extra_whitespace", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "bool", + "value": true + }, "code": { "advanced": true, "dynamic": true, @@ -1380,21 +1405,25 @@ "show": true, "title_case": false, "type": "code", - "value": "import re\n\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.helpers.data import data_to_text\nfrom langflow.io import DropdownInput, MessageTextInput, Output\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = \"Load and retrive data from specified URLs.\"\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=\"Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.\",\n options=[\"Text\", \"Raw HTML\"],\n value=\"Text\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Message\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.\n\n Raises an error if the string is not a valid URL.\n\n Parameters:\n string (str): The string to be checked and possibly modified.\n\n Returns:\n str: The modified string that is ensured to be a URL.\n\n Raises:\n ValueError: If the string is not a valid URL.\n \"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n # Basic URL validation regex\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\" # optional protocol\n r\"(www\\.)?\" # optional www\n r\"([a-zA-Z0-9.-]+)\" # domain\n r\"(\\.[a-zA-Z]{2,})?\" # top-level domain\n r\"(:\\d+)?\" # optional port\n r\"(\\/[^\\s]*)?$\", # optional path\n re.IGNORECASE,\n )\n\n if not url_regex.match(string):\n msg = f\"Invalid URL: {string}\"\n raise ValueError(msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n docs = loader.load()\n data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n self.status = data\n return data\n\n def fetch_content_text(self) -> Message:\n data = self.fetch_content()\n\n result_string = data_to_text(\"{text}\", data)\n self.status = result_string\n return Message(text=result_string)\n\n def as_dataframe(self) -> DataFrame:\n return DataFrame(self.fetch_content())\n" + "value": "import asyncio\nimport json\nimport re\n\nimport aiohttp\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = (\n \"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, \"\n \"or JSON, with options for cleaning and separating multiple outputs.\"\n )\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=(\n \"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML \"\n \"content, or 'JSON' to extract JSON from the HTML.\"\n ),\n options=[\"Text\", \"Raw HTML\", \"JSON\"],\n value=\"Text\",\n real_time_refresh=True,\n ),\n StrInput(\n name=\"separator\",\n display_name=\"Separator\",\n value=\"\\n\\n\",\n show=True,\n info=(\n \"Specify the separator to use between multiple outputs. Default for Text is '\\\\n\\\\n'. \"\n \"Default for Raw HTML is '\\\\n\\\\n'.\"\n ),\n ),\n BoolInput(\n name=\"clean_extra_whitespace\",\n display_name=\"Clean Extra Whitespace\",\n value=True,\n show=True,\n info=\"Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Text\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n async def validate_json_content(self, url: str) -> bool:\n \"\"\"Validates if the URL content is actually JSON.\"\"\"\n try:\n async with aiohttp.ClientSession() as session, session.get(url) as response:\n http_ok = 200\n if response.status != http_ok:\n return False\n\n content = await response.text()\n try:\n json.loads(content)\n except json.JSONDecodeError:\n return False\n else:\n return True\n except (aiohttp.ClientError, asyncio.TimeoutError):\n # Log specific error for debugging if needed\n return False\n\n def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:\n \"\"\"Dynamically update fields based on selected format.\"\"\"\n if field_name == \"format\":\n is_text_mode = field_value == \"Text\"\n is_json_mode = field_value == \"JSON\"\n build_config[\"separator\"][\"value\"] = \"\\n\\n\" if is_text_mode else \"\\n\\n\"\n build_config[\"clean_extra_whitespace\"][\"show\"] = is_text_mode\n build_config[\"separator\"][\"show\"] = not is_json_mode\n return build_config\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a valid URL.\"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\"\n r\"(www\\.)?\"\n r\"([a-zA-Z0-9.-]+)\"\n r\"(\\.[a-zA-Z]{2,})?\"\n r\"(:\\d+)?\"\n r\"(\\/[^\\s]*)?$\",\n re.IGNORECASE,\n )\n\n error_msg = \"Invalid URL - \" + string\n if not url_regex.match(string):\n raise ValueError(error_msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n \"\"\"Fetch content based on selected format.\"\"\"\n urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})\n\n no_urls_msg = \"No valid URLs provided.\"\n if not urls:\n raise ValueError(no_urls_msg)\n\n # If JSON format is selected, validate JSON content first\n if self.format == \"JSON\":\n for url in urls:\n is_json = asyncio.run(self.validate_json_content(url))\n if not is_json:\n error_msg = \"Invalid JSON content from URL - \" + url\n raise ValueError(error_msg)\n\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n\n docs = loader.load()\n\n if self.format == \"JSON\":\n data = []\n for doc in docs:\n try:\n json_content = json.loads(doc.page_content)\n data_dict = {\"text\": json.dumps(json_content, indent=2), **json_content, **doc.metadata}\n data.append(Data(**data_dict))\n except json.JSONDecodeError as err:\n source = doc.metadata.get(\"source\", \"unknown URL\")\n error_msg = \"Invalid JSON content from \" + source\n raise ValueError(error_msg) from err\n return data\n\n return [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n\n def fetch_content_text(self) -> Message:\n \"\"\"Fetch content and return as formatted text.\"\"\"\n data = self.fetch_content()\n\n if self.format == \"JSON\":\n text_list = [item.text for item in data]\n result = \"\\n\".join(text_list)\n else:\n text_list = [item.text for item in data]\n if self.format == \"Text\" and self.clean_extra_whitespace:\n text_list = [re.sub(r\"\\n{3,}\", \"\\n\\n\", text) for text in text_list]\n result = self.separator.join(text_list)\n\n self.status = result\n return Message(text=result)\n\n def as_dataframe(self) -> DataFrame:\n \"\"\"Return fetched content as a DataFrame.\"\"\"\n return DataFrame(self.fetch_content())\n" }, "format": { "_input_type": "DropdownInput", "advanced": false, "combobox": false, + "dialog_inputs": {}, "display_name": "Output Format", "dynamic": false, - "info": "Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", + "info": "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML content, or 'JSON' to extract JSON from the HTML.", "name": "format", "options": [ "Text", - "Raw HTML" + "Raw HTML", + "JSON" ], + "options_metadata": [], "placeholder": "", + "real_time_refresh": true, "required": false, "show": true, "title_case": false, @@ -1403,6 +1432,25 @@ "type": "str", "value": "Text" }, + "separator": { + "_input_type": "StrInput", + "advanced": false, + "display_name": "Separator", + "dynamic": false, + "info": "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. Default for Raw HTML is '\\n\\n'.", + "list": false, + "list_add_label": "Add More", + "load_from_db": false, + "name": "separator", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "str", + "value": "\n\n" + }, "tools_metadata": { "_input_type": "TableInput", "advanced": false, @@ -1438,37 +1486,43 @@ "table_schema": { "columns": [ { + "default": "None", "description": "Specify the name of the tool.", "disable_edit": false, "display_name": "Tool Name", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": false, "name": "name", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "Describe the purpose of the tool.", "disable_edit": false, "display_name": "Tool Description", "edit_mode": "popover", "filterable": false, "formatter": "text", + "hidden": false, "name": "description", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "The default identifiers for the tools and cannot be changed.", "disable_edit": true, "display_name": "Tool Identifiers", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": true, "name": "tags", "sortable": false, - "type": "text" + "type": "str" } ] }, @@ -1480,21 +1534,21 @@ "type": "table", "value": [ { - "description": "fetch_content() - Load and retrive data from specified URLs.", + "description": "fetch_content() - Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "name": "URL-fetch_content", "tags": [ "URL-fetch_content" ] }, { - "description": "fetch_content_text() - Load and retrive data from specified URLs.", + "description": "fetch_content_text() - Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "name": "URL-fetch_content_text", "tags": [ "URL-fetch_content_text" ] }, { - "description": "as_dataframe() - Load and retrive data from specified URLs.", + "description": "as_dataframe() - Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "name": "URL-as_dataframe", "tags": [ "URL-as_dataframe" @@ -1531,21 +1585,21 @@ "type": "URL" }, "dragging": false, - "id": "URL-yM3pY", + "id": "URL-BvxUK", "measured": { - "height": 403, - "width": 320 + "height": 660, + "width": 360 }, "position": { - "x": 1225.8773509111968, - "y": 27.333577318641687 + "x": 1236.8269016193576, + "y": -173.8644169438124 }, "selected": false, "type": "genericNode" }, { "data": { - "id": "note-Zo71g", + "id": "note-Xs1cm", "node": { "description": "# 📖 README\nRun an Agent with URL and Calculator tools available for its use. \nThe Agent decides which tool to use to solve a problem.\n## Quick start\n\n1. Add your OpenAI API key to the Agent.\n2. Open the Playground and chat with the Agent. Request some information about a recipe, and then ask to add two numbers together. In the responses, the Agent will use different tools to solve different problems.\n\n## Next steps\nConnect more tools to the Agent to create your perfect assistant.\n\nFor more, see the [Langflow docs](https://docs.langflow.org/agents-tool-calling-agent-component).", "display_name": "", @@ -1557,21 +1611,21 @@ "type": "note" }, "dragging": false, - "id": "note-Zo71g", + "id": "note-Xs1cm", "measured": { - "height": 324, - "width": 324 + "height": 668, + "width": 328 }, "position": { "x": 775.5268622081468, "y": 27.927425537464444 }, - "selected": true, + "selected": false, "type": "noteNode" }, { "data": { - "id": "note-enlN0", + "id": "note-t8RFb", "node": { "description": "### 💡 Add your OpenAI API key here👇", "display_name": "", @@ -1582,10 +1636,10 @@ }, "type": "note" }, - "id": "note-enlN0", + "id": "note-t8RFb", "measured": { - "height": 698, - "width": 324 + "height": 324, + "width": 326 }, "position": { "x": 1648.6876745095624, @@ -1596,7 +1650,7 @@ }, { "data": { - "id": "CalculatorComponent-zxpIu", + "id": "CalculatorComponent-GTSkO", "node": { "base_classes": [ "Data" @@ -1616,7 +1670,7 @@ "icon": "calculator", "key": "CalculatorComponent", "legacy": false, - "lf_version": "1.1.1", + "lf_version": "1.2.0", "metadata": {}, "minimized": false, "output_types": [], @@ -1630,6 +1684,7 @@ "name": "component_as_tool", "required_inputs": null, "selected": "Tool", + "tool_mode": true, "types": [ "Tool" ], @@ -1716,37 +1771,43 @@ "table_schema": { "columns": [ { + "default": "None", "description": "Specify the name of the tool.", "disable_edit": false, "display_name": "Tool Name", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": false, "name": "name", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "Describe the purpose of the tool.", "disable_edit": false, "display_name": "Tool Description", "edit_mode": "popover", "filterable": false, "formatter": "text", + "hidden": false, "name": "description", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "The default identifiers for the tools and cannot be changed.", "disable_edit": true, "display_name": "Tool Identifiers", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": true, "name": "tags", "sortable": false, - "type": "text" + "type": "str" } ] }, @@ -1773,10 +1834,10 @@ "type": "CalculatorComponent" }, "dragging": false, - "id": "CalculatorComponent-zxpIu", + "id": "CalculatorComponent-GTSkO", "measured": { - "height": 333, - "width": 320 + "height": 374, + "width": 360 }, "position": { "x": 1233.166256931297, @@ -1787,9 +1848,9 @@ } ], "viewport": { - "x": 528.5325055091846, - "y": 51.2172319875898, - "zoom": 0.12508906569071757 + "x": -373.5765333322679, + "y": 128.23283781763195, + "zoom": 0.6186520962774794 } }, "description": "A simple but powerful starter agent.", diff --git a/src/backend/base/langflow/initial_setup/starter_projects/Travel Planning Agents.json b/src/backend/base/langflow/initial_setup/starter_projects/Travel Planning Agents.json index f9099c775..d99a3e804 100644 --- a/src/backend/base/langflow/initial_setup/starter_projects/Travel Planning Agents.json +++ b/src/backend/base/langflow/initial_setup/starter_projects/Travel Planning Agents.json @@ -7,7 +7,7 @@ "data": { "sourceHandle": { "dataType": "Agent", - "id": "Agent-8PETJ", + "id": "Agent-e5naw", "name": "response", "output_types": [ "Message" @@ -15,7 +15,7 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "ChatOutput-1fycm", + "id": "ChatOutput-NdG8s", "inputTypes": [ "Data", "DataFrame", @@ -24,11 +24,11 @@ "type": "str" } }, - "id": "reactflow__edge-Agent-8PETJ{œdataTypeœ:œAgentœ,œidœ:œAgent-8PETJœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-ChatOutput-1fycm{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-1fycmœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", - "source": "Agent-8PETJ", - "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-8PETJœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", - "target": "ChatOutput-1fycm", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œChatOutput-1fycmœ, œinputTypesœ: [œDataœ, œDataFrameœ, œMessageœ], œtypeœ: œstrœ}" + "id": "reactflow__edge-Agent-e5naw{œdataTypeœ:œAgentœ,œidœ:œAgent-e5nawœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-ChatOutput-NdG8s{œfieldNameœ:œinput_valueœ,œidœ:œChatOutput-NdG8sœ,œinputTypesœ:[œDataœ,œDataFrameœ,œMessageœ],œtypeœ:œstrœ}", + "source": "Agent-e5naw", + "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-e5nawœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", + "target": "ChatOutput-NdG8s", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œChatOutput-NdG8sœ, œinputTypesœ: [œDataœ, œDataFrameœ, œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -36,7 +36,7 @@ "data": { "sourceHandle": { "dataType": "Agent", - "id": "Agent-Mb2Ep", + "id": "Agent-PWZOu", "name": "response", "output_types": [ "Message" @@ -44,18 +44,18 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "Agent-8PETJ", + "id": "Agent-e5naw", "inputTypes": [ "Message" ], "type": "str" } }, - "id": "reactflow__edge-Agent-Mb2Ep{œdataTypeœ:œAgentœ,œidœ:œAgent-Mb2Epœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-Agent-8PETJ{œfieldNameœ:œinput_valueœ,œidœ:œAgent-8PETJœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", - "source": "Agent-Mb2Ep", - "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-Mb2Epœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", - "target": "Agent-8PETJ", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-8PETJœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" + "id": "reactflow__edge-Agent-PWZOu{œdataTypeœ:œAgentœ,œidœ:œAgent-PWZOuœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-Agent-e5naw{œfieldNameœ:œinput_valueœ,œidœ:œAgent-e5nawœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", + "source": "Agent-PWZOu", + "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-PWZOuœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", + "target": "Agent-e5naw", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-e5nawœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -63,7 +63,7 @@ "data": { "sourceHandle": { "dataType": "Agent", - "id": "Agent-VEG7r", + "id": "Agent-3H5qa", "name": "response", "output_types": [ "Message" @@ -71,18 +71,18 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "Agent-Mb2Ep", + "id": "Agent-PWZOu", "inputTypes": [ "Message" ], "type": "str" } }, - "id": "reactflow__edge-Agent-VEG7r{œdataTypeœ:œAgentœ,œidœ:œAgent-VEG7rœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-Agent-Mb2Ep{œfieldNameœ:œinput_valueœ,œidœ:œAgent-Mb2Epœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", - "source": "Agent-VEG7r", - "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-VEG7rœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", - "target": "Agent-Mb2Ep", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-Mb2Epœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" + "id": "reactflow__edge-Agent-3H5qa{œdataTypeœ:œAgentœ,œidœ:œAgent-3H5qaœ,œnameœ:œresponseœ,œoutput_typesœ:[œMessageœ]}-Agent-PWZOu{œfieldNameœ:œinput_valueœ,œidœ:œAgent-PWZOuœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", + "source": "Agent-3H5qa", + "sourceHandle": "{œdataTypeœ: œAgentœ, œidœ: œAgent-3H5qaœ, œnameœ: œresponseœ, œoutput_typesœ: [œMessageœ]}", + "target": "Agent-PWZOu", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-PWZOuœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -90,7 +90,7 @@ "data": { "sourceHandle": { "dataType": "ChatInput", - "id": "ChatInput-Nbu0b", + "id": "ChatInput-LIxJY", "name": "message", "output_types": [ "Message" @@ -98,18 +98,18 @@ }, "targetHandle": { "fieldName": "input_value", - "id": "Agent-VEG7r", + "id": "Agent-3H5qa", "inputTypes": [ "Message" ], "type": "str" } }, - "id": "reactflow__edge-ChatInput-Nbu0b{œdataTypeœ:œChatInputœ,œidœ:œChatInput-Nbu0bœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Agent-VEG7r{œfieldNameœ:œinput_valueœ,œidœ:œAgent-VEG7rœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", - "source": "ChatInput-Nbu0b", - "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-Nbu0bœ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", - "target": "Agent-VEG7r", - "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-VEG7rœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" + "id": "reactflow__edge-ChatInput-LIxJY{œdataTypeœ:œChatInputœ,œidœ:œChatInput-LIxJYœ,œnameœ:œmessageœ,œoutput_typesœ:[œMessageœ]}-Agent-3H5qa{œfieldNameœ:œinput_valueœ,œidœ:œAgent-3H5qaœ,œinputTypesœ:[œMessageœ],œtypeœ:œstrœ}", + "source": "ChatInput-LIxJY", + "sourceHandle": "{œdataTypeœ: œChatInputœ, œidœ: œChatInput-LIxJYœ, œnameœ: œmessageœ, œoutput_typesœ: [œMessageœ]}", + "target": "Agent-3H5qa", + "targetHandle": "{œfieldNameœ: œinput_valueœ, œidœ: œAgent-3H5qaœ, œinputTypesœ: [œMessageœ], œtypeœ: œstrœ}" }, { "animated": false, @@ -117,7 +117,7 @@ "data": { "sourceHandle": { "dataType": "URL", - "id": "URL-hJKTA", + "id": "URL-XDhvs", "name": "component_as_tool", "output_types": [ "Tool" @@ -125,18 +125,18 @@ }, "targetHandle": { "fieldName": "tools", - "id": "Agent-Mb2Ep", + "id": "Agent-PWZOu", "inputTypes": [ "Tool" ], "type": "other" } }, - "id": "reactflow__edge-URL-hJKTA{œdataTypeœ:œURLœ,œidœ:œURL-hJKTAœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-Mb2Ep{œfieldNameœ:œtoolsœ,œidœ:œAgent-Mb2Epœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", - "source": "URL-hJKTA", - "sourceHandle": "{œdataTypeœ: œURLœ, œidœ: œURL-hJKTAœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", - "target": "Agent-Mb2Ep", - "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-Mb2Epœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" + "id": "reactflow__edge-URL-XDhvs{œdataTypeœ:œURLœ,œidœ:œURL-XDhvsœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-PWZOu{œfieldNameœ:œtoolsœ,œidœ:œAgent-PWZOuœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", + "source": "URL-XDhvs", + "sourceHandle": "{œdataTypeœ: œURLœ, œidœ: œURL-XDhvsœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", + "target": "Agent-PWZOu", + "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-PWZOuœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" }, { "animated": false, @@ -144,7 +144,7 @@ "data": { "sourceHandle": { "dataType": "CalculatorComponent", - "id": "CalculatorComponent-T0153", + "id": "CalculatorComponent-N83tJ", "name": "component_as_tool", "output_types": [ "Tool" @@ -152,18 +152,18 @@ }, "targetHandle": { "fieldName": "tools", - "id": "Agent-8PETJ", + "id": "Agent-e5naw", "inputTypes": [ "Tool" ], "type": "other" } }, - "id": "reactflow__edge-CalculatorComponent-T0153{œdataTypeœ:œCalculatorComponentœ,œidœ:œCalculatorComponent-T0153œ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-8PETJ{œfieldNameœ:œtoolsœ,œidœ:œAgent-8PETJœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", - "source": "CalculatorComponent-T0153", - "sourceHandle": "{œdataTypeœ: œCalculatorComponentœ, œidœ: œCalculatorComponent-T0153œ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", - "target": "Agent-8PETJ", - "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-8PETJœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" + "id": "reactflow__edge-CalculatorComponent-N83tJ{œdataTypeœ:œCalculatorComponentœ,œidœ:œCalculatorComponent-N83tJœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-e5naw{œfieldNameœ:œtoolsœ,œidœ:œAgent-e5nawœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", + "source": "CalculatorComponent-N83tJ", + "sourceHandle": "{œdataTypeœ: œCalculatorComponentœ, œidœ: œCalculatorComponent-N83tJœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", + "target": "Agent-e5naw", + "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-e5nawœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" }, { "animated": false, @@ -171,7 +171,7 @@ "data": { "sourceHandle": { "dataType": "SearchComponent", - "id": "SearchComponent-a9OCR", + "id": "SearchComponent-fF95b", "name": "component_as_tool", "output_types": [ "Tool" @@ -179,24 +179,24 @@ }, "targetHandle": { "fieldName": "tools", - "id": "Agent-VEG7r", + "id": "Agent-3H5qa", "inputTypes": [ "Tool" ], "type": "other" } }, - "id": "reactflow__edge-SearchComponent-a9OCR{œdataTypeœ:œSearchComponentœ,œidœ:œSearchComponent-a9OCRœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-VEG7r{œfieldNameœ:œtoolsœ,œidœ:œAgent-VEG7rœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", - "source": "SearchComponent-a9OCR", - "sourceHandle": "{œdataTypeœ: œSearchComponentœ, œidœ: œSearchComponent-a9OCRœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", - "target": "Agent-VEG7r", - "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-VEG7rœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" + "id": "reactflow__edge-SearchComponent-fF95b{œdataTypeœ:œSearchComponentœ,œidœ:œSearchComponent-fF95bœ,œnameœ:œcomponent_as_toolœ,œoutput_typesœ:[œToolœ]}-Agent-3H5qa{œfieldNameœ:œtoolsœ,œidœ:œAgent-3H5qaœ,œinputTypesœ:[œToolœ],œtypeœ:œotherœ}", + "source": "SearchComponent-fF95b", + "sourceHandle": "{œdataTypeœ: œSearchComponentœ, œidœ: œSearchComponent-fF95bœ, œnameœ: œcomponent_as_toolœ, œoutput_typesœ: [œToolœ]}", + "target": "Agent-3H5qa", + "targetHandle": "{œfieldNameœ: œtoolsœ, œidœ: œAgent-3H5qaœ, œinputTypesœ: [œToolœ], œtypeœ: œotherœ}" } ], "nodes": [ { "data": { - "id": "ChatInput-Nbu0b", + "id": "ChatInput-LIxJY", "node": { "base_classes": [ "Message" @@ -467,10 +467,10 @@ }, "dragging": false, "height": 262, - "id": "ChatInput-Nbu0b", + "id": "ChatInput-LIxJY", "measured": { "height": 262, - "width": 320 + "width": 360 }, "position": { "x": 1756.77096149088, @@ -488,7 +488,7 @@ "data": { "description": "Display a chat message in the Playground.", "display_name": "Chat Output", - "id": "ChatOutput-1fycm", + "id": "ChatOutput-NdG8s", "node": { "base_classes": [ "Message" @@ -770,10 +770,10 @@ }, "dragging": false, "height": 262, - "id": "ChatOutput-1fycm", + "id": "ChatOutput-NdG8s", "measured": { "height": 262, - "width": 320 + "width": 360 }, "position": { "x": 4349.229697347143, @@ -791,7 +791,7 @@ "data": { "description": "Define the agent's instructions, then enter a task to complete using tools.", "display_name": "City Selection Agent", - "id": "Agent-VEG7r", + "id": "Agent-3H5qa", "node": { "base_classes": [ "Message" @@ -1390,10 +1390,10 @@ }, "dragging": true, "height": 725, - "id": "Agent-VEG7r", + "id": "Agent-3H5qa", "measured": { "height": 725, - "width": 320 + "width": 360 }, "position": { "x": 2472.7748760933105, @@ -1411,7 +1411,7 @@ "data": { "description": "Define the agent's instructions, then enter a task to complete using tools.", "display_name": "Local Expert Agent", - "id": "Agent-Mb2Ep", + "id": "Agent-PWZOu", "node": { "base_classes": [ "Message" @@ -2010,10 +2010,10 @@ }, "dragging": false, "height": 725, - "id": "Agent-Mb2Ep", + "id": "Agent-PWZOu", "measured": { "height": 725, - "width": 320 + "width": 360 }, "position": { "x": 3185.66991544494, @@ -2031,7 +2031,7 @@ "data": { "description": "Define the agent's instructions, then enter a task to complete using tools.", "display_name": "Travel Concierge Agent", - "id": "Agent-8PETJ", + "id": "Agent-e5naw", "node": { "base_classes": [ "Message" @@ -2630,10 +2630,10 @@ }, "dragging": false, "height": 725, - "id": "Agent-8PETJ", + "id": "Agent-e5naw", "measured": { "height": 725, - "width": 320 + "width": 360 }, "position": { "x": 3889.695953842898, @@ -2649,7 +2649,7 @@ }, { "data": { - "id": "note-qGsTQ", + "id": "note-qLmmh", "node": { "description": "# Travel Planning Agents \n\nThe travel planning system is a smart setup that uses several specialized \"agents\" to help plan incredible trips. Imagine each agent as a travel expert focusing on a part of your journey. Here's how it works:\n\n- **User-Friendly Start:** You start by telling the system about your travel needs—where you want to go and what you love to do.\n\n- **Data Collection:** The agents uses its tools to gather current info about various destinations, like the best travel times, weather, and costs.\n\n- **Three Key Agents:**\n - **City Selection Agent:** Picks the best places to visit based on your likes and current data.\n - **Local Expert Agent:** Gathers interesting details about what to do and see in the chosen city.\n - **Travel Concierge Agent:** Builds a day-by-day plan that includes where to stay, eat, and explore!\n\n- **Tools and Data:** Each agent uses tools to find and organize the latest information so you get recommendations that are both accurate and exciting.\n\n- **Final Plan:** Once everything is put together, you receive a complete, easy-to-follow travel itinerary, perfect for your adventure!\n", "display_name": "", @@ -2660,10 +2660,10 @@ }, "dragging": false, "height": 603, - "id": "note-qGsTQ", + "id": "note-qLmmh", "measured": { "height": 603, - "width": 325 + "width": 328 }, "position": { "x": 1319.2860379588103, @@ -2674,7 +2674,7 @@ "y": 92.06058855045646 }, "resizing": false, - "selected": true, + "selected": false, "style": { "height": 636, "width": 324 @@ -2684,7 +2684,7 @@ }, { "data": { - "id": "note-8T0vy", + "id": "note-lZk3i", "node": { "description": "# **City Selection Agent**\n - **Purpose:** This agent evaluates potential travel destinations based on user input and external data sources.\n - **Core Functions:** Analyzes factors such as weather, local events, and travel costs to recommend optimal cities.\n - **Tools Utilized:** Employs APIs and data-fetching tools to gather real-time information for decision-making.\n", "display_name": "", @@ -2697,10 +2697,10 @@ }, "dragging": false, "height": 334, - "id": "note-8T0vy", + "id": "note-lZk3i", "measured": { "height": 334, - "width": 325 + "width": 328 }, "position": { "x": 2122.4146132377227, @@ -2721,7 +2721,7 @@ }, { "data": { - "id": "note-FZGo4", + "id": "note-twqCP", "node": { "description": "# **Local Expert Agent**\n - **Purpose:** Focused on gathering and providing an in-depth guide to the selected city.\n - **Core Functions:** Compiles insights into cultural attractions, local customs, and unique experiences.\n - **Tools Utilized:** Uses web content fetchers and data APIs to collect detailed local insights and enhance the user understanding with hidden gems.\n", "display_name": "", @@ -2734,10 +2734,10 @@ }, "dragging": false, "height": 342, - "id": "note-FZGo4", + "id": "note-twqCP", "measured": { "height": 342, - "width": 325 + "width": 328 }, "position": { "x": 2827.660803823376, @@ -2758,7 +2758,7 @@ }, { "data": { - "id": "note-FAKzZ", + "id": "note-bhmU0", "node": { "description": "# **Travel Concierge Agent**\n - **Purpose:** Crafts detailed travel itineraries that are customized to the traveler's interests and needs.\n - **Core Functions:** Offers a comprehensive daily schedule, including accommodations, dining spots, and activities.\n - **Tools Utilized:** Integrates calculators and data tools for accurate budget planning and itinerary logistics.", "display_name": "", @@ -2771,10 +2771,10 @@ }, "dragging": false, "height": 336, - "id": "note-FAKzZ", + "id": "note-bhmU0", "measured": { "height": 336, - "width": 325 + "width": 328 }, "position": { "x": 3536.084279543714, @@ -2795,7 +2795,7 @@ }, { "data": { - "id": "note-aM3pJ", + "id": "note-pFFw3", "node": { "description": "## Configure the agent by obtaining your OpenAI API key from [platform.openai.com](https://platform.openai.com). Under \"Model Provider\", choose:\n- OpenAI: Default, requires only API key\n- Anthropic/Azure/Groq/NVIDIA: Each requires their own API keys\n- Custom: Use your own model endpoint + authentication\n\nSelect model and input API key before running the flow.", "display_name": "", @@ -2808,10 +2808,10 @@ }, "dragging": false, "height": 324, - "id": "note-aM3pJ", + "id": "note-pFFw3", "measured": { "height": 324, - "width": 325 + "width": 328 }, "position": { "x": 2463.3881993480218, @@ -2828,7 +2828,9 @@ }, { "data": { - "id": "URL-hJKTA", + "description": "Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", + "display_name": "URL", + "id": "URL-XDhvs", "node": { "base_classes": [ "Data", @@ -2836,22 +2838,21 @@ "Message" ], "beta": false, - "category": "data", "conditional_paths": [], "custom_fields": {}, - "description": "Load and retrive data from specified URLs.", + "description": "Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "display_name": "URL", "documentation": "", "edited": false, "field_order": [ "urls", - "format" + "format", + "separator", + "clean_extra_whitespace" ], "frozen": false, "icon": "layout-template", - "key": "URL", "legacy": false, - "lf_version": "1.1.1", "metadata": {}, "minimized": false, "output_types": [], @@ -2865,6 +2866,7 @@ "name": "component_as_tool", "required_inputs": null, "selected": "Tool", + "tool_mode": true, "types": [ "Tool" ], @@ -2872,9 +2874,26 @@ } ], "pinned": false, - "score": 2.220446049250313e-16, "template": { "_type": "Component", + "clean_extra_whitespace": { + "_input_type": "BoolInput", + "advanced": false, + "display_name": "Clean Extra Whitespace", + "dynamic": false, + "info": "Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.", + "list": false, + "list_add_label": "Add More", + "name": "clean_extra_whitespace", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "bool", + "value": true + }, "code": { "advanced": true, "dynamic": true, @@ -2891,7 +2910,7 @@ "show": true, "title_case": false, "type": "code", - "value": "import re\n\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.helpers.data import data_to_text\nfrom langflow.io import DropdownInput, MessageTextInput, Output\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = \"Load and retrive data from specified URLs.\"\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=\"Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.\",\n options=[\"Text\", \"Raw HTML\"],\n value=\"Text\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Message\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.\n\n Raises an error if the string is not a valid URL.\n\n Parameters:\n string (str): The string to be checked and possibly modified.\n\n Returns:\n str: The modified string that is ensured to be a URL.\n\n Raises:\n ValueError: If the string is not a valid URL.\n \"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n # Basic URL validation regex\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\" # optional protocol\n r\"(www\\.)?\" # optional www\n r\"([a-zA-Z0-9.-]+)\" # domain\n r\"(\\.[a-zA-Z]{2,})?\" # top-level domain\n r\"(:\\d+)?\" # optional port\n r\"(\\/[^\\s]*)?$\", # optional path\n re.IGNORECASE,\n )\n\n if not url_regex.match(string):\n msg = f\"Invalid URL: {string}\"\n raise ValueError(msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n docs = loader.load()\n data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n self.status = data\n return data\n\n def fetch_content_text(self) -> Message:\n data = self.fetch_content()\n\n result_string = data_to_text(\"{text}\", data)\n self.status = result_string\n return Message(text=result_string)\n\n def as_dataframe(self) -> DataFrame:\n return DataFrame(self.fetch_content())\n" + "value": "import asyncio\nimport json\nimport re\n\nimport aiohttp\nfrom langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader\n\nfrom langflow.custom import Component\nfrom langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput\nfrom langflow.schema import Data\nfrom langflow.schema.dataframe import DataFrame\nfrom langflow.schema.message import Message\n\n\nclass URLComponent(Component):\n display_name = \"URL\"\n description = (\n \"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, \"\n \"or JSON, with options for cleaning and separating multiple outputs.\"\n )\n icon = \"layout-template\"\n name = \"URL\"\n\n inputs = [\n MessageTextInput(\n name=\"urls\",\n display_name=\"URLs\",\n is_list=True,\n tool_mode=True,\n placeholder=\"Enter a URL...\",\n list_add_label=\"Add URL\",\n ),\n DropdownInput(\n name=\"format\",\n display_name=\"Output Format\",\n info=(\n \"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML \"\n \"content, or 'JSON' to extract JSON from the HTML.\"\n ),\n options=[\"Text\", \"Raw HTML\", \"JSON\"],\n value=\"Text\",\n real_time_refresh=True,\n ),\n StrInput(\n name=\"separator\",\n display_name=\"Separator\",\n value=\"\\n\\n\",\n show=True,\n info=(\n \"Specify the separator to use between multiple outputs. Default for Text is '\\\\n\\\\n'. \"\n \"Default for Raw HTML is '\\\\n\\\\n'.\"\n ),\n ),\n BoolInput(\n name=\"clean_extra_whitespace\",\n display_name=\"Clean Extra Whitespace\",\n value=True,\n show=True,\n info=\"Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.\",\n ),\n ]\n\n outputs = [\n Output(display_name=\"Data\", name=\"data\", method=\"fetch_content\"),\n Output(display_name=\"Text\", name=\"text\", method=\"fetch_content_text\"),\n Output(display_name=\"DataFrame\", name=\"dataframe\", method=\"as_dataframe\"),\n ]\n\n async def validate_json_content(self, url: str) -> bool:\n \"\"\"Validates if the URL content is actually JSON.\"\"\"\n try:\n async with aiohttp.ClientSession() as session, session.get(url) as response:\n http_ok = 200\n if response.status != http_ok:\n return False\n\n content = await response.text()\n try:\n json.loads(content)\n except json.JSONDecodeError:\n return False\n else:\n return True\n except (aiohttp.ClientError, asyncio.TimeoutError):\n # Log specific error for debugging if needed\n return False\n\n def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:\n \"\"\"Dynamically update fields based on selected format.\"\"\"\n if field_name == \"format\":\n is_text_mode = field_value == \"Text\"\n is_json_mode = field_value == \"JSON\"\n build_config[\"separator\"][\"value\"] = \"\\n\\n\" if is_text_mode else \"\\n\\n\"\n build_config[\"clean_extra_whitespace\"][\"show\"] = is_text_mode\n build_config[\"separator\"][\"show\"] = not is_json_mode\n return build_config\n\n def ensure_url(self, string: str) -> str:\n \"\"\"Ensures the given string is a valid URL.\"\"\"\n if not string.startswith((\"http://\", \"https://\")):\n string = \"http://\" + string\n\n url_regex = re.compile(\n r\"^(https?:\\/\\/)?\"\n r\"(www\\.)?\"\n r\"([a-zA-Z0-9.-]+)\"\n r\"(\\.[a-zA-Z]{2,})?\"\n r\"(:\\d+)?\"\n r\"(\\/[^\\s]*)?$\",\n re.IGNORECASE,\n )\n\n error_msg = \"Invalid URL - \" + string\n if not url_regex.match(string):\n raise ValueError(error_msg)\n\n return string\n\n def fetch_content(self) -> list[Data]:\n \"\"\"Fetch content based on selected format.\"\"\"\n urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})\n\n no_urls_msg = \"No valid URLs provided.\"\n if not urls:\n raise ValueError(no_urls_msg)\n\n # If JSON format is selected, validate JSON content first\n if self.format == \"JSON\":\n for url in urls:\n is_json = asyncio.run(self.validate_json_content(url))\n if not is_json:\n error_msg = \"Invalid JSON content from URL - \" + url\n raise ValueError(error_msg)\n\n if self.format == \"Raw HTML\":\n loader = AsyncHtmlLoader(web_path=urls, encoding=\"utf-8\")\n else:\n loader = WebBaseLoader(web_paths=urls, encoding=\"utf-8\")\n\n docs = loader.load()\n\n if self.format == \"JSON\":\n data = []\n for doc in docs:\n try:\n json_content = json.loads(doc.page_content)\n data_dict = {\"text\": json.dumps(json_content, indent=2), **json_content, **doc.metadata}\n data.append(Data(**data_dict))\n except json.JSONDecodeError as err:\n source = doc.metadata.get(\"source\", \"unknown URL\")\n error_msg = \"Invalid JSON content from \" + source\n raise ValueError(error_msg) from err\n return data\n\n return [Data(text=doc.page_content, **doc.metadata) for doc in docs]\n\n def fetch_content_text(self) -> Message:\n \"\"\"Fetch content and return as formatted text.\"\"\"\n data = self.fetch_content()\n\n if self.format == \"JSON\":\n text_list = [item.text for item in data]\n result = \"\\n\".join(text_list)\n else:\n text_list = [item.text for item in data]\n if self.format == \"Text\" and self.clean_extra_whitespace:\n text_list = [re.sub(r\"\\n{3,}\", \"\\n\\n\", text) for text in text_list]\n result = self.separator.join(text_list)\n\n self.status = result\n return Message(text=result)\n\n def as_dataframe(self) -> DataFrame:\n \"\"\"Return fetched content as a DataFrame.\"\"\"\n return DataFrame(self.fetch_content())\n" }, "format": { "_input_type": "DropdownInput", @@ -2900,14 +2919,16 @@ "dialog_inputs": {}, "display_name": "Output Format", "dynamic": false, - "info": "Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.", + "info": "Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML content, or 'JSON' to extract JSON from the HTML.", "name": "format", "options": [ "Text", - "Raw HTML" + "Raw HTML", + "JSON" ], "options_metadata": [], "placeholder": "", + "real_time_refresh": true, "required": false, "show": true, "title_case": false, @@ -2916,6 +2937,25 @@ "type": "str", "value": "Text" }, + "separator": { + "_input_type": "StrInput", + "advanced": false, + "display_name": "Separator", + "dynamic": false, + "info": "Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. Default for Raw HTML is '\\n\\n'.", + "list": false, + "list_add_label": "Add More", + "load_from_db": false, + "name": "separator", + "placeholder": "", + "required": false, + "show": true, + "title_case": false, + "tool_mode": false, + "trace_as_metadata": true, + "type": "str", + "value": "langflow" + }, "tools_metadata": { "_input_type": "TableInput", "advanced": false, @@ -2951,37 +2991,43 @@ "table_schema": { "columns": [ { + "default": "None", "description": "Specify the name of the tool.", "disable_edit": false, "display_name": "Tool Name", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": false, "name": "name", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "Describe the purpose of the tool.", "disable_edit": false, "display_name": "Tool Description", "edit_mode": "popover", "filterable": false, "formatter": "text", + "hidden": false, "name": "description", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "The default identifiers for the tools and cannot be changed.", "disable_edit": true, "display_name": "Tool Identifiers", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": true, "name": "tags", "sortable": false, - "type": "text" + "type": "str" } ] }, @@ -2993,21 +3039,21 @@ "type": "table", "value": [ { - "description": "fetch_content() - Load and retrive data from specified URLs.", + "description": "fetch_content() - Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "name": "URL-fetch_content", "tags": [ "URL-fetch_content" ] }, { - "description": "fetch_content_text() - Load and retrive data from specified URLs.", + "description": "fetch_content_text() - Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "name": "URL-fetch_content_text", "tags": [ "URL-fetch_content_text" ] }, { - "description": "as_dataframe() - Load and retrive data from specified URLs.", + "description": "as_dataframe() - Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, or JSON, with options for cleaning and separating multiple outputs.", "name": "URL-as_dataframe", "tags": [ "URL-as_dataframe" @@ -3045,10 +3091,10 @@ "type": "URL" }, "dragging": false, - "id": "URL-hJKTA", + "id": "URL-XDhvs", "measured": { - "height": 403, - "width": 320 + "height": 660, + "width": 360 }, "position": { "x": 2829.4526852839367, @@ -3059,7 +3105,7 @@ }, { "data": { - "id": "CalculatorComponent-T0153", + "id": "CalculatorComponent-N83tJ", "node": { "base_classes": [ "Data" @@ -3236,10 +3282,10 @@ "type": "CalculatorComponent" }, "dragging": false, - "id": "CalculatorComponent-T0153", + "id": "CalculatorComponent-N83tJ", "measured": { - "height": 333, - "width": 320 + "height": 374, + "width": 360 }, "position": { "x": 3540.356346381247, @@ -3250,7 +3296,7 @@ }, { "data": { - "id": "SearchComponent-a9OCR", + "id": "SearchComponent-fF95b", "node": { "base_classes": [ "Data", @@ -3290,6 +3336,7 @@ "name": "component_as_tool", "required_inputs": null, "selected": "Tool", + "tool_mode": true, "types": [ "Tool" ], @@ -3474,37 +3521,43 @@ "table_schema": { "columns": [ { + "default": "None", "description": "Specify the name of the tool.", "disable_edit": false, "display_name": "Tool Name", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": false, "name": "name", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "Describe the purpose of the tool.", "disable_edit": false, "display_name": "Tool Description", "edit_mode": "popover", "filterable": false, "formatter": "text", + "hidden": false, "name": "description", "sortable": false, - "type": "text" + "type": "str" }, { + "default": "None", "description": "The default identifiers for the tools and cannot be changed.", "disable_edit": true, "display_name": "Tool Identifiers", "edit_mode": "inline", "filterable": false, "formatter": "text", + "hidden": true, "name": "tags", "sortable": false, - "type": "text" + "type": "str" } ] }, @@ -3538,10 +3591,10 @@ "type": "SearchComponent" }, "dragging": false, - "id": "SearchComponent-a9OCR", + "id": "SearchComponent-fF95b", "measured": { - "height": 477, - "width": 320 + "height": 536, + "width": 360 }, "position": { "x": 2089.0393126914205, @@ -3552,9 +3605,9 @@ } ], "viewport": { - "x": -452.59655672949, - "y": 325.42241573046994, - "zoom": 0.392950977749874 + "x": -473.30523608095336, + "y": 166.93251319338754, + "zoom": 0.41164535038506667 } }, "description": "Create a travel planning chatbot that uses specialized agents to craft personalized trip itineraries.", diff --git a/src/backend/tests/unit/components/data/test_url_component.py b/src/backend/tests/unit/components/data/test_url_component.py index 52e4a672b..a520b7a36 100644 --- a/src/backend/tests/unit/components/data/test_url_component.py +++ b/src/backend/tests/unit/components/data/test_url_component.py @@ -111,7 +111,7 @@ class TestURLComponent(ComponentTestBaseWithoutClient): component.set_attributes({"urls": ["not_a_valid_url"]}) # Test that invalid URLs raise a ValueError - with pytest.raises(ValueError, match="Invalid URL: http://not_a_valid_url"): + with pytest.raises(ValueError, match="Invalid URL - http://not_a_valid_url"): component.fetch_content() def test_url_component_multiple_urls(self, mock_web_load): diff --git a/src/backend/tests/unit/test_database.py b/src/backend/tests/unit/test_database.py index 9c2485b1f..54d00b977 100644 --- a/src/backend/tests/unit/test_database.py +++ b/src/backend/tests/unit/test_database.py @@ -181,12 +181,15 @@ async def test_read_flows_components_only_paginated(client: AsyncClient, logged_ FlowCreate(name=f"Flow {i}", description="description", data={}, is_component=True) for i in range(number_of_flows) ] + for flow in flows: response = await client.post("api/v1/flows/", json=flow.model_dump(), headers=logged_in_headers) assert response.status_code == 201 + response = await client.get( "api/v1/flows/", headers=logged_in_headers, params={"components_only": True, "get_all": False} ) + assert response.status_code == 200 response_json = response.json() assert response_json["total"] == 10 diff --git a/src/frontend/tests/assets/test_audio_file.wav b/src/frontend/tests/assets/test_audio_file.wav index abd61d7d296163b15fb285f89bc21cd56f92adba..3a4c8b974478506e468157592f6b6aac9ebce3c2 100644 GIT binary patch delta 169 zcmWN=M+(9K002Sk8hcCZy?1};VZlqjB6tw|LHw3GFuzxlO8g*2nhaTTh9qTr%d0Yi_vZj(Z+>