Refactor components and update JSON output for clarity
This commit is contained in:
parent
29a31083e5
commit
002430100c
4 changed files with 33 additions and 8 deletions
|
|
@ -60,7 +60,7 @@ class URLComponent(Component):
|
||||||
|
|
||||||
def fetch_content(self) -> Data:
|
def fetch_content(self) -> Data:
|
||||||
urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]
|
urls = [self.ensure_url(url.strip()) for url in self.urls if url.strip()]
|
||||||
loader = WebBaseLoader(web_paths=urls)
|
loader = WebBaseLoader(web_paths=urls, encoding="utf-8")
|
||||||
docs = loader.load()
|
docs = loader.load()
|
||||||
data = [Data(content=doc.page_content, **doc.metadata) for doc in docs]
|
data = [Data(content=doc.page_content, **doc.metadata) for doc in docs]
|
||||||
self.status = data
|
self.status = data
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ from langflow.custom import CustomComponent
|
||||||
from langflow.schema import Data
|
from langflow.schema import Data
|
||||||
|
|
||||||
|
|
||||||
class ExtractKeyFromdataComponent(CustomComponent):
|
class ExtractKeyFromDataComponent(CustomComponent):
|
||||||
display_name = "Extract Key From Data"
|
display_name = "Extract Key From Data"
|
||||||
description = "Extracts a key from a data."
|
description = "Extracts a key from a data."
|
||||||
beta: bool = True
|
beta: bool = True
|
||||||
|
|
@ -9,7 +9,7 @@ from langflow.utils.util import unescape_string
|
||||||
|
|
||||||
class SplitContentComponent(Component):
|
class SplitContentComponent(Component):
|
||||||
display_name: str = "Split Content"
|
display_name: str = "Split Content"
|
||||||
description: str = "Split textual content into chunks of a specified length."
|
description: str = "Split textual content into chunks based on specified criteria."
|
||||||
icon = "split"
|
icon = "split"
|
||||||
|
|
||||||
inputs = [
|
inputs = [
|
||||||
|
|
@ -35,7 +35,21 @@ class SplitContentComponent(Component):
|
||||||
IntInput(
|
IntInput(
|
||||||
name="chunk_size",
|
name="chunk_size",
|
||||||
display_name="Chunk Size",
|
display_name="Chunk Size",
|
||||||
info="The maximum length (in number of characters) of each chunk. Defaults to 0 (no chunking).",
|
info="The target length (in number of characters) of each chunk.",
|
||||||
|
value=0,
|
||||||
|
advanced=True
|
||||||
|
),
|
||||||
|
IntInput(
|
||||||
|
name="min_chunk_size",
|
||||||
|
display_name="Minimum Chunk Size",
|
||||||
|
info="The minimum size of chunks. Smaller chunks will be merged.",
|
||||||
|
value=0,
|
||||||
|
advanced=True
|
||||||
|
),
|
||||||
|
IntInput(
|
||||||
|
name="max_chunk_size",
|
||||||
|
display_name="Maximum Chunk Size",
|
||||||
|
info="The maximum size of chunks. Larger chunks will be split.",
|
||||||
value=0,
|
value=0,
|
||||||
advanced=True
|
advanced=True
|
||||||
),
|
),
|
||||||
|
|
@ -46,12 +60,14 @@ class SplitContentComponent(Component):
|
||||||
]
|
]
|
||||||
|
|
||||||
def split_text(self) -> List[Data]:
|
def split_text(self) -> List[Data]:
|
||||||
# TODO: Add minimum chunk size - can be removed or merged
|
|
||||||
data = self.data if isinstance(self.data, list) else [self.data]
|
data = self.data if isinstance(self.data, list) else [self.data]
|
||||||
content_key = self.content_key
|
content_key = self.content_key
|
||||||
separator = unescape_string(self.separator)
|
separator = unescape_string(self.separator)
|
||||||
chunk_size = self.chunk_size
|
chunk_size = self.chunk_size
|
||||||
|
min_chunk_size = self.min_chunk_size
|
||||||
|
max_chunk_size = self.max_chunk_size
|
||||||
results = []
|
results = []
|
||||||
|
buffer = ""
|
||||||
|
|
||||||
for row in data:
|
for row in data:
|
||||||
content = row.data.get(content_key, '')
|
content = row.data.get(content_key, '')
|
||||||
|
|
@ -61,8 +77,17 @@ class SplitContentComponent(Component):
|
||||||
chunks = content.split(separator)
|
chunks = content.split(separator)
|
||||||
|
|
||||||
for chunk in chunks:
|
for chunk in chunks:
|
||||||
if chunk.strip():
|
buffer += chunk
|
||||||
results.append(Data(data={"parent": content, "text": chunk}))
|
while len(buffer) >= max_chunk_size:
|
||||||
|
results.append(Data(data={"parent": content, "chunk": buffer[:max_chunk_size]}))
|
||||||
|
buffer = buffer[max_chunk_size:]
|
||||||
|
if len(buffer) >= min_chunk_size:
|
||||||
|
results.append(Data(data={"parent": content, "chunk": buffer}))
|
||||||
|
buffer = ""
|
||||||
|
|
||||||
|
# Handle any remaining content that may not meet the min_chunk_size requirement
|
||||||
|
if buffer:
|
||||||
|
results.append(Data(data={"parent": content, "chunk": buffer}))
|
||||||
|
|
||||||
self.status = results
|
self.status = results
|
||||||
return results
|
return results
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
from .AgentComponent import AgentComponent
|
from .AgentComponent import AgentComponent
|
||||||
from .ClearMessageHistory import ClearMessageHistoryComponent
|
from .ClearMessageHistory import ClearMessageHistoryComponent
|
||||||
from .ExtractDataFromData import ExtractKeyFromDataComponent
|
from .ExtractKeyFromData import ExtractKeyFromDataComponent
|
||||||
from .FlowTool import FlowToolComponent
|
from .FlowTool import FlowToolComponent
|
||||||
from .Listen import ListenComponent
|
from .Listen import ListenComponent
|
||||||
from .ListFlows import ListFlowsComponent
|
from .ListFlows import ListFlowsComponent
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue