feat: enhance YouTubeTranscripts component with Data output support (#6113)
* 📝 (youtube_transcripts.py): update description of YouTubeTranscriptsComponent to be more concise and accurate ✨ (youtube_transcripts.py): add new output option 'data_output' to provide transcript along with the source video URL 🔧 (youtube_transcripts.py): add method 'get_data_output' to handle the new 'data_output' output option and return a Data object with transcript, video URL, and error message * [autofix.ci] apply automated fixes * 📝 (youtube_transcripts.py): improve documentation for get_data_output method to provide a clear description of the returned data object and its contents 🐛 (youtube_transcripts.py): handle specific exceptions from the youtube_transcript_api library to provide more informative error messages and improve error handling in the get_data_output method * [autofix.ci] apply automated fixes * 🐛 (youtube_transcripts.py): handle case where no transcripts are found by updating the error message and returning a default data object 🔧 (youtube_transcripts.py): refactor get_data_output method to use a default data object and combine all transcript parts into a single continuous text * [autofix.ci] apply automated fixes * ✨ (test_youtube_transcript_component.py): Add unit tests for YouTubeTranscriptsComponent to test various functionalities such as component initialization, output generation, error handling, and setting translation languages. * [autofix.ci] apply automated fixes * ✅ (test_youtube_transcript_component.py): update file_names_mapping fixture to return a non-empty list to properly test different versions of file names mapping in the YouTube transcripts component * [autofix.ci] apply automated fixes * 📝 (test_youtube_transcript_component.py): Add docstrings and improve variable names for better readability and maintainability 🔧 (test_youtube_transcript_component.py): Refactor error handling in test methods to use descriptive error messages and improve code readability * [autofix.ci] apply automated fixes --------- Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
This commit is contained in:
parent
17f1ecf997
commit
d98d37778d
4 changed files with 200 additions and 2 deletions
|
|
@ -5,7 +5,7 @@ from langchain_community.document_loaders.youtube import TranscriptFormat
|
|||
|
||||
from langflow.custom import Component
|
||||
from langflow.inputs import DropdownInput, IntInput, MultilineInput
|
||||
from langflow.schema import DataFrame, Message
|
||||
from langflow.schema import Data, DataFrame, Message
|
||||
from langflow.template import Output
|
||||
|
||||
|
||||
|
|
@ -13,7 +13,7 @@ class YouTubeTranscriptsComponent(Component):
|
|||
"""A component that extracts spoken content from YouTube videos as transcripts."""
|
||||
|
||||
display_name: str = "YouTube Transcripts"
|
||||
description: str = "Extracts spoken content from YouTube videos with both DataFrame and text output options."
|
||||
description: str = "Extracts spoken content from YouTube videos with multiple output options."
|
||||
icon: str = "YouTube"
|
||||
name = "YouTubeTranscripts"
|
||||
|
||||
|
|
@ -43,6 +43,7 @@ class YouTubeTranscriptsComponent(Component):
|
|||
outputs = [
|
||||
Output(name="dataframe", display_name="Chunks", method="get_dataframe_output"),
|
||||
Output(name="message", display_name="Transcript", method="get_message_output"),
|
||||
Output(name="data_output", display_name="Transcript + Source", method="get_data_output"),
|
||||
]
|
||||
|
||||
def _load_transcripts(self, *, as_chunks: bool = True):
|
||||
|
|
@ -68,6 +69,7 @@ class YouTubeTranscriptsComponent(Component):
|
|||
start_seconds %= 60
|
||||
timestamp = f"{start_minutes:02d}:{start_seconds:02d}"
|
||||
data.append({"timestamp": timestamp, "text": doc.page_content})
|
||||
|
||||
return DataFrame(pd.DataFrame(data))
|
||||
|
||||
except (youtube_transcript_api.TranscriptsDisabled, youtube_transcript_api.NoTranscriptFound) as exc:
|
||||
|
|
@ -83,3 +85,32 @@ class YouTubeTranscriptsComponent(Component):
|
|||
except (youtube_transcript_api.TranscriptsDisabled, youtube_transcript_api.NoTranscriptFound) as exc:
|
||||
error_msg = f"Failed to get YouTube transcripts: {exc!s}"
|
||||
return Message(text=error_msg)
|
||||
|
||||
def get_data_output(self) -> Data:
|
||||
"""Creates a structured data object with transcript and metadata.
|
||||
|
||||
Returns a Data object containing transcript text, video URL, and any error
|
||||
messages that occurred during processing. The object includes:
|
||||
- 'transcript': continuous text from the entire video (concatenated if multiple parts)
|
||||
- 'video_url': the input YouTube URL
|
||||
- 'error': error message if an exception occurs
|
||||
"""
|
||||
default_data = {"transcript": "", "video_url": self.url, "error": None}
|
||||
|
||||
try:
|
||||
transcripts = self._load_transcripts(as_chunks=False)
|
||||
if not transcripts:
|
||||
default_data["error"] = "No transcripts found."
|
||||
return Data(data=default_data)
|
||||
|
||||
# Combine all transcript parts
|
||||
full_transcript = " ".join(doc.page_content for doc in transcripts)
|
||||
return Data(data={"transcript": full_transcript, "video_url": self.url})
|
||||
|
||||
except (
|
||||
youtube_transcript_api.TranscriptsDisabled,
|
||||
youtube_transcript_api.NoTranscriptFound,
|
||||
youtube_transcript_api.CouldNotRetrieveTranscript,
|
||||
) as exc:
|
||||
default_data["error"] = str(exc)
|
||||
return Data(data=default_data)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue