feat: Add JSON field extraction and enhanced URL validation (#6051)
* URL component improvement - JSON URL * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * ♻️ (url.py): refactor URLComponent class to simplify data_dict creation by using dictionary unpacking instead of manual key-value pairs * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * 📝 (url.py): improve formatting of info string for DropdownInput in URLComponent class ♻️ (url.py): refactor ensure_url method to simplify logic and improve readability 🐛 (url.py): fix error handling in URLComponent class for invalid JSON content * [autofix.ci] apply automated fixes * ✨ (url.py): Add BoolInput and StrInput to support new features in URLComponent 📝 (url.py): Update description in URLComponent to provide more detailed information about its functionality ♻️ (url.py): Refactor update_build_config method in URLComponent to dynamically update fields based on selected format 🐛 (url.py): Fix ensure_url method in URLComponent to ensure valid URLs are provided and handle exceptions properly 🐛 (url.py): Fix fetch_content method in URLComponent to handle cases where no valid URLs are provided and improve error handling 🐛 (url.py): Fix fetch_content_text method in URLComponent to correctly format and clean text output based on selected format and settings 🐛 (url.py): Fix as_dataframe method in URLComponent to return fetched content as a DataFrame object * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * ♻️ (url.py): remove unnecessary comments and improve code readability by removing redundant comments and adjusting code structure. * [autofix.ci] apply automated fixes * 📝 (url.py): improve readability by splitting long description and info strings into multiple lines 🐛 (url.py): handle cases where invalid URLs or JSON URLs are provided, and provide informative error messages 🐛 (url.py): handle cases where no valid URLs are provided and raise an error with a clear message * 🔧 (Blog Writer.json, Custom Component Maker.json, Graph Vector Store RAG.json): resolve merge conflicts in JSON files related to the 'format' field options to ensure consistency across starter projects. * [autofix.ci] apply automated fixes * 🐛 (url.py): fix validation of JSON content from URLs to ensure correct handling of JSON data ✨ (url.py): introduce async validation of JSON content from URLs using aiohttp to improve performance and reliability * [autofix.ci] apply automated fixes * [autofix.ci] apply automated fixes (attempt 2/3) * merge fix * ✅ (test_audio_file.wav): update test_audio_file.wav to fix binary file differences in the test asset * 🐛 (test_url_component.py): update error message format to improve clarity and consistency * update templates * 🐛 (test_database.py): fix error handling in test_read_flows_components_only_paginated to properly catch and log exceptions during test execution * 📝 (backend): Add noqa comments to files to ignore specific linting rule A005 🔧 (test_database.py): Remove duplicate import statement for sqlalchemy ♻️ (test_database.py): Refactor test_read_flows_components_only_paginated function for better readability and maintainability * 🐛 (test_database.py): fix test_read_flows_components_only_paginated to handle exceptions and provide more context in case of failure * 📝 (test_database.py): remove unnecessary comment to improve code readability and maintainability * 📝 (backend): Remove unnecessary noqa comments from __init__.py files 🔧 (test_database.py): Refactor test_read_flows_components_only_paginated function for better readability and maintainability * [autofix.ci] apply automated fixes --------- Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
This commit is contained in:
parent
c41eaf0071
commit
47753d37d3
10 changed files with 638 additions and 252 deletions
|
|
@ -1,10 +1,12 @@
|
|||
import asyncio
|
||||
import json
|
||||
import re
|
||||
|
||||
import aiohttp
|
||||
from langchain_community.document_loaders import AsyncHtmlLoader, WebBaseLoader
|
||||
|
||||
from langflow.custom import Component
|
||||
from langflow.helpers.data import data_to_text
|
||||
from langflow.io import DropdownInput, MessageTextInput, Output
|
||||
from langflow.io import BoolInput, DropdownInput, MessageTextInput, Output, StrInput
|
||||
from langflow.schema import Data
|
||||
from langflow.schema.dataframe import DataFrame
|
||||
from langflow.schema.message import Message
|
||||
|
|
@ -12,7 +14,10 @@ from langflow.schema.message import Message
|
|||
|
||||
class URLComponent(Component):
|
||||
display_name = "URL"
|
||||
description = "Load and retrive data from specified URLs."
|
||||
description = (
|
||||
"Load and retrieve data from specified URLs. Supports output in plain text, raw HTML, "
|
||||
"or JSON, with options for cleaning and separating multiple outputs."
|
||||
)
|
||||
icon = "layout-template"
|
||||
name = "URL"
|
||||
|
||||
|
|
@ -28,69 +33,143 @@ class URLComponent(Component):
|
|||
DropdownInput(
|
||||
name="format",
|
||||
display_name="Output Format",
|
||||
info="Output Format. Use 'Text' to extract the text from the HTML or 'Raw HTML' for the raw HTML content.",
|
||||
options=["Text", "Raw HTML"],
|
||||
info=(
|
||||
"Output Format. Use 'Text' to extract text from the HTML, 'Raw HTML' for the raw HTML "
|
||||
"content, or 'JSON' to extract JSON from the HTML."
|
||||
),
|
||||
options=["Text", "Raw HTML", "JSON"],
|
||||
value="Text",
|
||||
real_time_refresh=True,
|
||||
),
|
||||
StrInput(
|
||||
name="separator",
|
||||
display_name="Separator",
|
||||
value="\n\n",
|
||||
show=True,
|
||||
info=(
|
||||
"Specify the separator to use between multiple outputs. Default for Text is '\\n\\n'. "
|
||||
"Default for Raw HTML is '\\n<!-- Separator -->\\n'."
|
||||
),
|
||||
),
|
||||
BoolInput(
|
||||
name="clean_extra_whitespace",
|
||||
display_name="Clean Extra Whitespace",
|
||||
value=True,
|
||||
show=True,
|
||||
info="Whether to clean excessive blank lines in the text output. Only applies to 'Text' format.",
|
||||
),
|
||||
]
|
||||
|
||||
outputs = [
|
||||
Output(display_name="Data", name="data", method="fetch_content"),
|
||||
Output(display_name="Message", name="text", method="fetch_content_text"),
|
||||
Output(display_name="Text", name="text", method="fetch_content_text"),
|
||||
Output(display_name="DataFrame", name="dataframe", method="as_dataframe"),
|
||||
]
|
||||
|
||||
async def validate_json_content(self, url: str) -> bool:
|
||||
"""Validates if the URL content is actually JSON."""
|
||||
try:
|
||||
async with aiohttp.ClientSession() as session, session.get(url) as response:
|
||||
http_ok = 200
|
||||
if response.status != http_ok:
|
||||
return False
|
||||
|
||||
content = await response.text()
|
||||
try:
|
||||
json.loads(content)
|
||||
except json.JSONDecodeError:
|
||||
return False
|
||||
else:
|
||||
return True
|
||||
except (aiohttp.ClientError, asyncio.TimeoutError):
|
||||
# Log specific error for debugging if needed
|
||||
return False
|
||||
|
||||
def update_build_config(self, build_config: dict, field_value: str, field_name: str | None = None) -> dict:
|
||||
"""Dynamically update fields based on selected format."""
|
||||
if field_name == "format":
|
||||
is_text_mode = field_value == "Text"
|
||||
is_json_mode = field_value == "JSON"
|
||||
build_config["separator"]["value"] = "\n\n" if is_text_mode else "\n<!-- Separator -->\n"
|
||||
build_config["clean_extra_whitespace"]["show"] = is_text_mode
|
||||
build_config["separator"]["show"] = not is_json_mode
|
||||
return build_config
|
||||
|
||||
def ensure_url(self, string: str) -> str:
|
||||
"""Ensures the given string is a URL by adding 'http://' if it doesn't start with 'http://' or 'https://'.
|
||||
|
||||
Raises an error if the string is not a valid URL.
|
||||
|
||||
Parameters:
|
||||
string (str): The string to be checked and possibly modified.
|
||||
|
||||
Returns:
|
||||
str: The modified string that is ensured to be a URL.
|
||||
|
||||
Raises:
|
||||
ValueError: If the string is not a valid URL.
|
||||
"""
|
||||
"""Ensures the given string is a valid URL."""
|
||||
if not string.startswith(("http://", "https://")):
|
||||
string = "http://" + string
|
||||
|
||||
# Basic URL validation regex
|
||||
url_regex = re.compile(
|
||||
r"^(https?:\/\/)?" # optional protocol
|
||||
r"(www\.)?" # optional www
|
||||
r"([a-zA-Z0-9.-]+)" # domain
|
||||
r"(\.[a-zA-Z]{2,})?" # top-level domain
|
||||
r"(:\d+)?" # optional port
|
||||
r"(\/[^\s]*)?$", # optional path
|
||||
r"^(https?:\/\/)?"
|
||||
r"(www\.)?"
|
||||
r"([a-zA-Z0-9.-]+)"
|
||||
r"(\.[a-zA-Z]{2,})?"
|
||||
r"(:\d+)?"
|
||||
r"(\/[^\s]*)?$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
error_msg = "Invalid URL - " + string
|
||||
if not url_regex.match(string):
|
||||
msg = f"Invalid URL: {string}"
|
||||
raise ValueError(msg)
|
||||
raise ValueError(error_msg)
|
||||
|
||||
return string
|
||||
|
||||
def fetch_content(self) -> list[Data]:
|
||||
urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]
|
||||
"""Fetch content based on selected format."""
|
||||
urls = list({self.ensure_url(url.strip()) for url in self.urls if url.strip()})
|
||||
|
||||
no_urls_msg = "No valid URLs provided."
|
||||
if not urls:
|
||||
raise ValueError(no_urls_msg)
|
||||
|
||||
# If JSON format is selected, validate JSON content first
|
||||
if self.format == "JSON":
|
||||
for url in urls:
|
||||
is_json = asyncio.run(self.validate_json_content(url))
|
||||
if not is_json:
|
||||
error_msg = "Invalid JSON content from URL - " + url
|
||||
raise ValueError(error_msg)
|
||||
|
||||
if self.format == "Raw HTML":
|
||||
loader = AsyncHtmlLoader(web_path=urls, encoding="utf-8")
|
||||
else:
|
||||
loader = WebBaseLoader(web_paths=urls, encoding="utf-8")
|
||||
|
||||
docs = loader.load()
|
||||
data = [Data(text=doc.page_content, **doc.metadata) for doc in docs]
|
||||
self.status = data
|
||||
return data
|
||||
|
||||
if self.format == "JSON":
|
||||
data = []
|
||||
for doc in docs:
|
||||
try:
|
||||
json_content = json.loads(doc.page_content)
|
||||
data_dict = {"text": json.dumps(json_content, indent=2), **json_content, **doc.metadata}
|
||||
data.append(Data(**data_dict))
|
||||
except json.JSONDecodeError as err:
|
||||
source = doc.metadata.get("source", "unknown URL")
|
||||
error_msg = "Invalid JSON content from " + source
|
||||
raise ValueError(error_msg) from err
|
||||
return data
|
||||
|
||||
return [Data(text=doc.page_content, **doc.metadata) for doc in docs]
|
||||
|
||||
def fetch_content_text(self) -> Message:
|
||||
"""Fetch content and return as formatted text."""
|
||||
data = self.fetch_content()
|
||||
|
||||
result_string = data_to_text("{text}", data)
|
||||
self.status = result_string
|
||||
return Message(text=result_string)
|
||||
if self.format == "JSON":
|
||||
text_list = [item.text for item in data]
|
||||
result = "\n".join(text_list)
|
||||
else:
|
||||
text_list = [item.text for item in data]
|
||||
if self.format == "Text" and self.clean_extra_whitespace:
|
||||
text_list = [re.sub(r"\n{3,}", "\n\n", text) for text in text_list]
|
||||
result = self.separator.join(text_list)
|
||||
|
||||
self.status = result
|
||||
return Message(text=result)
|
||||
|
||||
def as_dataframe(self) -> DataFrame:
|
||||
"""Return fetched content as a DataFrame."""
|
||||
return DataFrame(self.fetch_content())
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -111,7 +111,7 @@ class TestURLComponent(ComponentTestBaseWithoutClient):
|
|||
component.set_attributes({"urls": ["not_a_valid_url"]})
|
||||
|
||||
# Test that invalid URLs raise a ValueError
|
||||
with pytest.raises(ValueError, match="Invalid URL: http://not_a_valid_url"):
|
||||
with pytest.raises(ValueError, match="Invalid URL - http://not_a_valid_url"):
|
||||
component.fetch_content()
|
||||
|
||||
def test_url_component_multiple_urls(self, mock_web_load):
|
||||
|
|
|
|||
|
|
@ -181,12 +181,15 @@ async def test_read_flows_components_only_paginated(client: AsyncClient, logged_
|
|||
FlowCreate(name=f"Flow {i}", description="description", data={}, is_component=True)
|
||||
for i in range(number_of_flows)
|
||||
]
|
||||
|
||||
for flow in flows:
|
||||
response = await client.post("api/v1/flows/", json=flow.model_dump(), headers=logged_in_headers)
|
||||
assert response.status_code == 201
|
||||
|
||||
response = await client.get(
|
||||
"api/v1/flows/", headers=logged_in_headers, params={"components_only": True, "get_all": False}
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
response_json = response.json()
|
||||
assert response_json["total"] == 10
|
||||
|
|
|
|||
Binary file not shown.
|
|
@ -237,6 +237,6 @@ test(
|
|||
|
||||
// Count occurrences of modified_value in output
|
||||
const matches = output?.match(/modified_value/g) || [];
|
||||
expect(matches).toHaveLength(2);
|
||||
expect(matches).toHaveLength(1);
|
||||
},
|
||||
);
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue