feat: Add new Atlassian Confluence Component for document loading and vector database integration (#2718)
* feat: Add Gemma 2 to Groq model list (#2586) Add gemma2 to groq_constants.py * Adds new ConfluenceComponent module with lazy loading support - Implements ConfluenceComponent to load documents from the Confluence platform. - Adds necessary inputs, including URL, username, API key, space_key, and more. - Supports configuration of max_pages for pagination control. - Implements lazy loading in the load_documents method for incremental document processing. - Allows immediate processing of documents as they are loaded. This new module facilitates integration with the Confluence platform and enables efficient handling of large volumes of data. * Adds new ConfluenceComponent module - Implements ConfluenceComponent to load documents from the Confluence platform. - Adds necessary inputs, including URL, username, API key, space key, and more. - Supports configuration of max_pages for pagination control. This new module facilitates integration with the Confluence platform. * Updated load_documents method to use Data.from_document - Changed load_documents method to convert documents using Data..from_document instead of docs_to_data for better integration with Data module. - Updated trace_type to "tool" because the LangSmith API only supports one of the following types: ["tool", "chain", "llm", "retriever", "embedding", "prompt", "parser"]. * [autofix.ci] apply automated fixes --------- Co-authored-by: Gordon Stein <7331488+gsteinLTU@users.noreply.github.com> Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
This commit is contained in:
parent
b281a7d25e
commit
114cdb9ac6
6 changed files with 128 additions and 0 deletions
|
|
@ -0,0 +1,85 @@
|
|||
from typing import List
|
||||
from langflow.custom import Component
|
||||
from langflow.io import StrInput, SecretStrInput, BoolInput, DropdownInput, Output, IntInput
|
||||
from langflow.schema import Data
|
||||
from langchain_community.document_loaders import ConfluenceLoader
|
||||
from langchain_community.document_loaders.confluence import ContentFormat
|
||||
|
||||
|
||||
class ConfluenceComponent(Component):
|
||||
display_name = "Confluence"
|
||||
description = "Confluence wiki collaboration platform"
|
||||
documentation = "https://python.langchain.com/v0.2/docs/integrations/document_loaders/confluence/"
|
||||
trace_type = "tool"
|
||||
icon = "Confluence"
|
||||
name = "Confluence"
|
||||
|
||||
inputs = [
|
||||
StrInput(
|
||||
name="url",
|
||||
display_name="Site URL",
|
||||
required=True,
|
||||
info="The base URL of the Confluence Space. Example: https://<company>.atlassian.net/wiki.",
|
||||
),
|
||||
StrInput(
|
||||
name="username",
|
||||
display_name="Username",
|
||||
required=True,
|
||||
info="Atlassian User E-mail. Example: email@example.com",
|
||||
),
|
||||
SecretStrInput(
|
||||
name="api_key",
|
||||
display_name="API Key",
|
||||
required=True,
|
||||
info="Atlassian Key. Create at: https://id.atlassian.com/manage-profile/security/api-tokens",
|
||||
),
|
||||
StrInput(name="space_key", display_name="Space Key", required=True),
|
||||
BoolInput(name="cloud", display_name="Use Cloud?", required=True, value=True, advanced=True),
|
||||
DropdownInput(
|
||||
name="content_format",
|
||||
display_name="Content Format",
|
||||
options=[
|
||||
ContentFormat.EDITOR.value,
|
||||
ContentFormat.EXPORT_VIEW.value,
|
||||
ContentFormat.ANONYMOUS_EXPORT_VIEW.value,
|
||||
ContentFormat.STORAGE.value,
|
||||
ContentFormat.VIEW.value,
|
||||
],
|
||||
value=ContentFormat.STORAGE.value,
|
||||
required=True,
|
||||
advanced=True,
|
||||
info="Specify content format, defaults to ContentFormat.STORAGE",
|
||||
),
|
||||
IntInput(
|
||||
name="max_pages",
|
||||
display_name="Max Pages",
|
||||
required=False,
|
||||
value=1000,
|
||||
advanced=True,
|
||||
info="Maximum number of pages to retrieve in total, defaults 1000",
|
||||
),
|
||||
]
|
||||
|
||||
outputs = [
|
||||
Output(name="data", display_name="Data", method="load_documents"),
|
||||
]
|
||||
|
||||
def build_confluence(self) -> ConfluenceLoader:
|
||||
content_format = ContentFormat(self.content_format)
|
||||
loader = ConfluenceLoader(
|
||||
url=self.url,
|
||||
username=self.username,
|
||||
api_key=self.api_key,
|
||||
cloud=self.cloud,
|
||||
space_key=self.space_key,
|
||||
content_format=content_format,
|
||||
max_pages=self.max_pages,
|
||||
)
|
||||
return loader
|
||||
|
||||
def load_documents(self) -> List[Data]:
|
||||
confluence = self.build_confluence()
|
||||
documents = confluence.load()
|
||||
data = [Data.from_document(doc) for doc in documents] # Using the from_document method of Data
|
||||
self.status = data
|
||||
return data
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
from .Confluence import ConfluenceComponent
|
||||
|
||||
__all__ = ["ConfluenceComponent"]
|
||||
Loading…
Add table
Add a link
Reference in a new issue