refactor vector stores

This commit is contained in:
Rodrigo 2024-06-17 11:36:32 -03:00
commit fa7707c148
13 changed files with 1442 additions and 1046 deletions

View file

@ -1,115 +1,169 @@
from typing import List, Optional, Union from langflow.custom import Component
from langflow.field_typing import Text
from langchain_core.retrievers import BaseRetriever from langflow.inputs import (
StrInput,
from langflow.custom import CustomComponent IntInput,
from langflow.field_typing import Embeddings, VectorStore BoolInput,
DropdownInput,
MultilineInput,
HandleInput,
)
from langflow.schema import Data from langflow.schema import Data
from langflow.template import Output
from langflow.field_typing import Embeddings
from loguru import logger
class AstraDBVectorStoreComponent(CustomComponent): class AstraDBComponent(Component):
display_name = "Astra DB" display_name: str = "Astra DB Vector Store"
description = "Builds or loads an Astra DB Vector Store." description: str = "Implementation of Vector Store using Astra DB with search capabilities"
icon = "AstraDB" documentation: str = "https://python.langchain.com/docs/integrations/vectorstores/astradb"
field_order = ["token", "api_endpoint", "collection_name", "inputs", "embedding"] icon: str = "AstraDB"
def build_config(self): inputs = [
return { StrInput(
"inputs": { name="collection_name",
"display_name": "Inputs", display_name="Collection Name",
"info": "Optional list of data to be processed and stored in the vector store.", info="The name of the collection within Astra DB where the vectors will be stored.",
}, ),
"embedding": {"display_name": "Embedding", "info": "Embedding to use"}, StrInput(
"collection_name": { name="token",
"display_name": "Collection Name", display_name="Astra DB Application Token",
"info": "The name of the collection within Astra DB where the vectors will be stored.", info="Authentication token for accessing Astra DB.",
}, password=True,
"token": { ),
"display_name": "Astra DB Application Token", StrInput(
"info": "Authentication token for accessing Astra DB.", name="api_endpoint",
"password": True, display_name="API Endpoint",
}, info="API endpoint URL for the Astra DB service.",
"api_endpoint": { ),
"display_name": "API Endpoint", StrInput(
"info": "API endpoint URL for the Astra DB service.", name="code",
}, display_name="Code",
"namespace": { advanced=True,
"display_name": "Namespace", ),
"info": "Optional namespace within Astra DB to use for the collection.", HandleInput(
"advanced": True, name="vector_store_inputs",
}, display_name="Vector Store Inputs",
"metric": { input_types=["Document", "Data"],
"display_name": "Metric", is_list=True,
"info": "Optional distance metric for vector comparisons in the vector store.", ),
"advanced": True, HandleInput(
}, name="embedding",
"batch_size": { display_name="Embedding",
"display_name": "Batch Size", input_types=["Embeddings"],
"info": "Optional number of data to process in a single batch.", ),
"advanced": True, StrInput(
}, name="namespace",
"bulk_insert_batch_concurrency": { display_name="Namespace",
"display_name": "Bulk Insert Batch Concurrency", info="Optional namespace within Astra DB to use for the collection.",
"info": "Optional concurrency level for bulk insert operations.", advanced=True,
"advanced": True, ),
}, DropdownInput(
"bulk_insert_overwrite_concurrency": { name="metric",
"display_name": "Bulk Insert Overwrite Concurrency", display_name="Metric",
"info": "Optional concurrency level for bulk insert operations that overwrite existing data.", info="Optional distance metric for vector comparisons in the vector store.",
"advanced": True, options=["cosine", "dot_product", "euclidean"],
}, advanced=True,
"bulk_delete_concurrency": { ),
"display_name": "Bulk Delete Concurrency", IntInput(
"info": "Optional concurrency level for bulk delete operations.", name="batch_size",
"advanced": True, display_name="Batch Size",
}, info="Optional number of data to process in a single batch.",
"setup_mode": { advanced=True,
"display_name": "Setup Mode", ),
"info": "Configuration mode for setting up the vector store, with options like “Sync”, “Async”, or “Off”.", IntInput(
"options": ["Sync", "Async", "Off"], name="bulk_insert_batch_concurrency",
"advanced": True, display_name="Bulk Insert Batch Concurrency",
}, info="Optional concurrency level for bulk insert operations.",
"pre_delete_collection": { advanced=True,
"display_name": "Pre Delete Collection", ),
"info": "Boolean flag to determine whether to delete the collection before creating a new one.", IntInput(
"advanced": True, name="bulk_insert_overwrite_concurrency",
}, display_name="Bulk Insert Overwrite Concurrency",
"metadata_indexing_include": { info="Optional concurrency level for bulk insert operations that overwrite existing data.",
"display_name": "Metadata Indexing Include", advanced=True,
"info": "Optional list of metadata fields to include in the indexing.", ),
"advanced": True, IntInput(
}, name="bulk_delete_concurrency",
"metadata_indexing_exclude": { display_name="Bulk Delete Concurrency",
"display_name": "Metadata Indexing Exclude", info="Optional concurrency level for bulk delete operations.",
"info": "Optional list of metadata fields to exclude from the indexing.", advanced=True,
"advanced": True, ),
}, DropdownInput(
"collection_indexing_policy": { name="setup_mode",
"display_name": "Collection Indexing Policy", display_name="Setup Mode",
"info": "Optional dictionary defining the indexing policy for the collection.", info="Configuration mode for setting up the vector store, with options like 'Sync', 'Async', or 'Off'.",
"advanced": True, options=["Sync", "Async", "Off"],
}, advanced=True,
} ),
BoolInput(
name="pre_delete_collection",
display_name="Pre Delete Collection",
info="Boolean flag to determine whether to delete the collection before creating a new one.",
advanced=True,
),
StrInput(
name="metadata_indexing_include",
display_name="Metadata Indexing Include",
info="Optional list of metadata fields to include in the indexing.",
advanced=True,
),
StrInput(
name="metadata_indexing_exclude",
display_name="Metadata Indexing Exclude",
info="Optional list of metadata fields to exclude from the indexing.",
advanced=True,
),
StrInput(
name="collection_indexing_policy",
display_name="Collection Indexing Policy",
info="Optional dictionary defining the indexing policy for the collection.",
advanced=True,
),
BoolInput(
name="add_to_vector_store",
display_name="Add to Vector Store",
info="If true, the Vector Store Inputs will be added to the Vector Store.",
),
MultilineInput(
name="search_input",
display_name="Search Input",
),
DropdownInput(
name="search_type",
display_name="Search Type",
options=["Similarity", "MMR"],
value="Similarity",
),
IntInput(
name="number_of_results",
display_name="Number of Results",
info="Number of results to return.",
advanced=True,
value=4,
),
]
def build( outputs = [
self, Output(
embedding: Embeddings, display_name="Vector Store",
token: str, name="vector_store",
api_endpoint: str, method="build_vector_store",
collection_name: str, ),
inputs: Optional[List[Data]] = None, Output(
namespace: Optional[str] = None, display_name="Base Retriever",
metric: Optional[str] = None, name="base_retriever",
batch_size: Optional[int] = None, method="build_base_retriever",
bulk_insert_batch_concurrency: Optional[int] = None, ),
bulk_insert_overwrite_concurrency: Optional[int] = None, Output(
bulk_delete_concurrency: Optional[int] = None, display_name="Search Results",
setup_mode: str = "Sync", name="search_results",
pre_delete_collection: bool = False, method="search_documents",
metadata_indexing_include: Optional[List[str]] = None, ),
metadata_indexing_exclude: Optional[List[str]] = None, ]
collection_indexing_policy: Optional[dict] = None,
) -> Union[VectorStore, BaseRetriever]: def build_vector_store(self):
try: try:
from langchain_astradb import AstraDBVectorStore from langchain_astradb import AstraDBVectorStore
from langchain_astradb.utils.astradb import SetupMode from langchain_astradb.utils.astradb import SetupMode
@ -120,47 +174,105 @@ class AstraDBVectorStoreComponent(CustomComponent):
) )
try: try:
setup_mode_value = SetupMode[setup_mode.upper()] setup_mode_value = SetupMode[self.setup_mode.upper()]
except KeyError: except KeyError:
raise ValueError(f"Invalid setup mode: {setup_mode}") raise ValueError(f"Invalid setup mode: {self.setup_mode}")
if inputs:
documents = [_input.to_lc_document() for _input in inputs]
vector_store = AstraDBVectorStore.from_documents( vector_store_kwargs = {
documents=documents, "embedding": self.embedding,
embedding=embedding, "collection_name": self.collection_name,
collection_name=collection_name, "token": self.token,
token=token, "api_endpoint": self.api_endpoint,
api_endpoint=api_endpoint, "namespace": self.namespace,
namespace=namespace, "metric": self.metric,
metric=metric, "batch_size": self.batch_size,
batch_size=batch_size, "bulk_insert_batch_concurrency": self.bulk_insert_batch_concurrency,
bulk_insert_batch_concurrency=bulk_insert_batch_concurrency, "bulk_insert_overwrite_concurrency": self.bulk_insert_overwrite_concurrency,
bulk_insert_overwrite_concurrency=bulk_insert_overwrite_concurrency, "bulk_delete_concurrency": self.bulk_delete_concurrency,
bulk_delete_concurrency=bulk_delete_concurrency, "setup_mode": setup_mode_value,
setup_mode=setup_mode_value, "pre_delete_collection": self.pre_delete_collection,
pre_delete_collection=pre_delete_collection, }
metadata_indexing_include=metadata_indexing_include,
metadata_indexing_exclude=metadata_indexing_exclude, if self.metadata_indexing_include:
collection_indexing_policy=collection_indexing_policy, vector_store_kwargs["metadata_indexing_include"] = self.metadata_indexing_include
elif self.metadata_indexing_exclude:
vector_store_kwargs["metadata_indexing_exclude"] = self.metadata_indexing_exclude
elif self.collection_indexing_policy:
vector_store_kwargs["collection_indexing_policy"] = self.collection_indexing_policy
try:
vector_store = AstraDBVectorStore(**vector_store_kwargs)
except Exception as e:
raise ValueError(f"Error initializing AstraDBVectorStore: {str(e)}") from e
if self.add_to_vector_store:
self._add_documents_to_vector_store(vector_store)
self.status = self._astradb_collection_to_data(vector_store.collection)
return vector_store
def build_base_retriever(self):
return self.build_vector_store()
def _add_documents_to_vector_store(self, vector_store):
documents = []
for _input in self.vector_store_inputs or []:
if isinstance(_input, Data):
documents.append(_input.to_lc_document())
else:
raise ValueError("Vector Store Inputs must be Data objects.")
if documents and self.embedding is not None:
logger.debug(f"Adding {len(documents)} documents to the Vector Store.")
try:
vector_store.add_documents(documents)
except Exception as e:
raise ValueError(f"Error adding documents to AstraDBVectorStore: {str(e)}") from e
else:
logger.debug("No documents to add to the Vector Store.")
def search_documents(self):
vector_store = self.build_vector_store()
logger.debug(f"Search input: {self.search_input}")
logger.debug(f"Search type: {self.search_type}")
logger.debug(f"Number of results: {self.number_of_results}")
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
try:
if self.search_type == "Similarity":
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
elif self.search_type == "MMR":
docs = vector_store.max_marginal_relevance_search(
query=self.search_input,
k=self.number_of_results,
) )
else: else:
vector_store = AstraDBVectorStore( raise ValueError(f"Invalid search type: {self.search_type}")
embedding=embedding, except Exception as e:
collection_name=collection_name, raise ValueError(f"Error performing search in AstraDBVectorStore: {str(e)}") from e
token=token,
api_endpoint=api_endpoint,
namespace=namespace,
metric=metric,
batch_size=batch_size,
bulk_insert_batch_concurrency=bulk_insert_batch_concurrency,
bulk_insert_overwrite_concurrency=bulk_insert_overwrite_concurrency,
bulk_delete_concurrency=bulk_delete_concurrency,
setup_mode=setup_mode_value,
pre_delete_collection=pre_delete_collection,
metadata_indexing_include=metadata_indexing_include,
metadata_indexing_exclude=metadata_indexing_exclude,
collection_indexing_policy=collection_indexing_policy,
)
return vector_store logger.debug(f"Retrieved documents: {len(docs)}")
data = self._docs_to_data(docs)
logger.debug(f"Converted documents to data: {len(data)}")
self.status = data
return data
else:
logger.debug("No search input provided. Skipping search.")
return []
def _astradb_collection_to_data(self, collection):
data = []
for item in collection["data"]:
data.append(Data(content=item["content"]))
return data
def _docs_to_data(self, docs):
data = []
for doc in docs:
data.append(Data(content=doc.page_content))
return data

View file

@ -2,79 +2,51 @@ from typing import Any, List, Optional, Tuple
from langchain_community.utilities.cassandra import SetupMode from langchain_community.utilities.cassandra import SetupMode
from langchain_community.vectorstores import Cassandra from langchain_community.vectorstores import Cassandra
from langchain.embeddings.base import Embeddings
from langchain.schema import BaseRetriever
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings, VectorStore from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput, DropdownInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class CassandraVectorStoreComponent(Component):
class CassandraVectorStoreComponent(CustomComponent):
display_name = "Cassandra" display_name = "Cassandra"
description = "Builds or loads a Cassandra Vector Store." description = "Cassandra Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/cassandra"
icon = "Cassandra" icon = "Cassandra"
field_order = ["token", "database_id", "table_name", "inputs", "embedding"]
def build_config(self): inputs = [
return { StrInput(name="token", display_name="Token", info="Authentication token for accessing Cassandra on Astra DB.", password=True, required=True),
"inputs": { StrInput(name="database_id", display_name="Database ID", info="The Astra database ID.", required=True),
"display_name": "Inputs", StrInput(name="table_name", display_name="Table Name", info="The name of the table where vectors will be stored.", required=True),
"info": "Optional list of data to be processed and stored in the vector store.", StrInput(name="keyspace", display_name="Keyspace", info="Optional key space within Astra DB. The keyspace should already be created.", advanced=True),
}, IntInput(name="ttl_seconds", display_name="TTL Seconds", info="Optional time-to-live for the added texts.", advanced=True),
"embedding": {"display_name": "Embedding", "info": "Embedding to use"}, IntInput(name="batch_size", display_name="Batch Size", info="Optional number of data to process in a single batch.", value=16, advanced=True),
"token": { StrInput(name="body_index_options", display_name="Body Index Options", info="Optional options used to create the body index.", advanced=True),
"display_name": "Token", DropdownInput(name="setup_mode", display_name="Setup Mode", info="Configuration mode for setting up the Cassandra table, with options like 'Sync', 'Async', or 'Off'.", options=["Sync", "Async", "Off"], value="Sync", advanced=True),
"info": "Authentication token for accessing Cassandra on Astra DB.", HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
"password": True, HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
}, BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
"database_id": { StrInput(name="search_input", display_name="Search Input"),
"display_name": "Database ID", IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
"info": "The Astra database ID.", ]
},
"table_name": {
"display_name": "Table Name",
"info": "The name of the table where vectors will be stored.",
},
"keyspace": {
"display_name": "Keyspace",
"info": "Optional key space within Astra DB. The keyspace should already be created.",
"advanced": True,
},
"ttl_seconds": {
"display_name": "TTL Seconds",
"info": "Optional time-to-live for the added texts.",
"advanced": True,
},
"batch_size": {
"display_name": "Batch Size",
"info": "Optional number of data to process in a single batch.",
"advanced": True,
},
"body_index_options": {
"display_name": "Body Index Options",
"info": "Optional options used to create the body index.",
"advanced": True,
},
"setup_mode": {
"display_name": "Setup Mode",
"info": "Configuration mode for setting up the Cassandra table, with options like 'Sync', 'Async', or 'Off'.",
"options": ["Sync", "Async", "Off"],
"advanced": True,
},
}
def build( outputs = [
self, Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=Cassandra),
embedding: Embeddings, Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
token: str, Output(display_name="Search Results", name="search_results", method="search_documents"),
database_id: str, ]
inputs: Optional[List[Data]] = None,
keyspace: Optional[str] = None, def build_vector_store(self) -> Cassandra:
table_name: str = "", return self._build_cassandra()
ttl_seconds: Optional[int] = None,
batch_size: int = 16, def build_base_retriever(self) -> BaseRetriever:
body_index_options: Optional[List[Tuple[str, Any]]] = None, return self._build_cassandra()
setup_mode: SetupMode = SetupMode.SYNC,
) -> VectorStore: def _build_cassandra(self) -> Cassandra:
try: try:
import cassio import cassio
except ImportError: except ImportError:
@ -83,29 +55,68 @@ class CassandraVectorStoreComponent(CustomComponent):
) )
cassio.init( cassio.init(
database_id=database_id, database_id=self.database_id,
token=token, token=self.token,
) )
if inputs: if self.add_to_vector_store:
documents = [_input.to_lc_document() for _input in inputs] documents = []
for _input in self.vector_store_inputs or []:
if isinstance(_input, Data):
documents.append(_input.to_lc_document())
else:
documents.append(_input)
if documents:
table = Cassandra.from_documents( table = Cassandra.from_documents(
documents=documents, documents=documents,
embedding=embedding, embedding=self.embedding,
table_name=table_name, table_name=self.table_name,
keyspace=keyspace, keyspace=self.keyspace,
ttl_seconds=ttl_seconds, ttl_seconds=self.ttl_seconds,
batch_size=batch_size, batch_size=self.batch_size,
body_index_options=body_index_options, body_index_options=self.body_index_options,
) )
else: else:
table = Cassandra( table = Cassandra(
embedding=embedding, embedding=self.embedding,
table_name=table_name, table_name=self.table_name,
keyspace=keyspace, keyspace=self.keyspace,
ttl_seconds=ttl_seconds, ttl_seconds=self.ttl_seconds,
body_index_options=body_index_options, body_index_options=self.body_index_options,
setup_mode=setup_mode, setup_mode=self.setup_mode,
)
else:
table = Cassandra(
embedding=self.embedding,
table_name=self.table_name,
keyspace=self.keyspace,
ttl_seconds=self.ttl_seconds,
body_index_options=self.body_index_options,
setup_mode=self.setup_mode,
) )
return table return table
def search_documents(self) -> List[Data]:
vector_store = self._build_cassandra()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
try:
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
except KeyError as e:
if "content" in str(e):
raise ValueError(
"You should ingest data through Langflow (or LangChain) to query it in Langflow. Your collection does not contain a field name 'content'."
)
else:
raise e
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -3,116 +3,182 @@ from typing import List, Optional, Union
import chromadb import chromadb
from chromadb.config import Settings from chromadb.config import Settings
from langchain_chroma import Chroma from langchain.vectorstores import Chroma
from langchain_core.embeddings import Embeddings from langchain.embeddings.base import Embeddings
from langchain_core.retrievers import BaseRetriever from langchain.schema import BaseRetriever
from langchain_core.vectorstores import VectorStore
from langflow.base.vectorstores.utils import chroma_collection_to_data from langflow.base.vectorstores.utils import chroma_collection_to_data
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput, DropdownInput
from langflow.schema import Data from langflow.schema import Data
from langflow.template import Output
from langflow.field_typing import Embeddings
from langflow.helpers.data import docs_to_data
from loguru import logger
class ChromaComponent(CustomComponent): class ChromaVectorStoreComponent(Component):
""" """
A custom component for implementing a Vector Store using Chroma. Chroma Vector Store with search capabilities
""" """
display_name: str = "Chroma" display_name: str = "Chroma DB"
description: str = "Implementation of Vector Store using Chroma" description: str = "Chroma Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/integrations/vectorstores/chroma" documentation = "https://python.langchain.com/docs/integrations/vectorstores/chroma"
icon = "Chroma" icon = "Chroma"
def build_config(self): inputs = [
StrInput(
name="collection_name",
display_name="Collection Name",
value="langflow",
),
StrInput(
name="persist_directory",
display_name="Persist Directory",
),
StrInput(
name="code",
display_name="Code",
advanced=True,
),
StrInput(
name="vector_store_inputs",
display_name="Vector Store Inputs",
input_types=["Document", "Data"],
is_list=True
),
HandleInput(
name="embedding",
display_name="Embedding",
input_types=["Embeddings"]
),
StrInput(
name="chroma_server_cors_allow_origins",
display_name="Server CORS Allow Origins",
advanced=True,
),
StrInput(
name="chroma_server_host",
display_name="Server Host",
advanced=True,
),
IntInput(
name="chroma_server_http_port",
display_name="Server HTTP Port",
advanced=True,
),
IntInput(
name="chroma_server_grpc_port",
display_name="Server gRPC Port",
advanced=True,
),
BoolInput(
name="chroma_server_ssl_enabled",
display_name="Server SSL Enabled",
advanced=True,
),
BoolInput(
name="allow_duplicates",
display_name="Allow Duplicates",
advanced=True,
info="If false, will not add documents that are already in the Vector Store.",
),
BoolInput(
name="add_to_vector_store",
display_name="Add to Vector Store",
info="If true, the Vector Store Inputs will be added to the Vector Store.",
),
StrInput(
name="search_input",
display_name="Search Input",
),
DropdownInput(
name="search_type",
display_name="Search Type",
options=["Similarity", "MMR"],
value="Similarity",
),
IntInput(
name="number_of_results",
display_name="Number of Results",
info="Number of results to return.",
advanced=True,
value=4,
),
]
outputs = [
Output(
display_name="Vector Store",
name="vector_store",
method="build_vector_store",
),
Output(
display_name="Base Retriever",
name="base_retriever",
method="build_base_retriever",
),
Output(
display_name="Search Results",
name="search_results",
method="search_documents",
),
]
def build_vector_store(self) -> Chroma:
""" """
Builds the configuration for the component. Builds the Vector Store object.
Returns:
- dict: A dictionary containing the configuration options for the component.
""" """
return { return self._build_chroma()
"collection_name": {"display_name": "Collection Name", "value": "langflow"},
"index_directory": {"display_name": "Persist Directory"},
"code": {"advanced": True, "display_name": "Code"},
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]},
"embedding": {"display_name": "Embedding"},
"chroma_server_cors_allow_origins": {
"display_name": "Server CORS Allow Origins",
"advanced": True,
},
"chroma_server_host": {"display_name": "Server Host", "advanced": True},
"chroma_server_http_port": {"display_name": "Server HTTP Port", "advanced": True},
"chroma_server_grpc_port": {
"display_name": "Server gRPC Port",
"advanced": True,
},
"chroma_server_ssl_enabled": {
"display_name": "Server SSL Enabled",
"advanced": True,
},
"allow_duplicates": {
"display_name": "Allow Duplicates",
"advanced": True,
"info": "If false, will not add documents that are already in the Vector Store.",
},
}
def build( def build_base_retriever(self) -> BaseRetriever:
self,
collection_name: str,
embedding: Embeddings,
chroma_server_ssl_enabled: bool,
index_directory: Optional[str] = None,
inputs: Optional[List[Data]] = None,
chroma_server_cors_allow_origins: List[str] = [],
chroma_server_host: Optional[str] = None,
chroma_server_http_port: Optional[int] = None,
chroma_server_grpc_port: Optional[int] = None,
allow_duplicates: bool = False,
) -> Union[VectorStore, BaseRetriever]:
""" """
Builds the Vector Store or BaseRetriever object. Builds the BaseRetriever object.
Args:
- collection_name (str): The name of the collection.
- embedding (Embeddings): The embeddings to use for the Vector Store.
- chroma_server_ssl_enabled (bool): Whether to enable SSL for the Chroma server.
- index_directory (Optional[str]): The directory to persist the Vector Store to.
- inputs (Optional[List[Data]]): The input data to use for the Vector Store.
- chroma_server_cors_allow_origins (List[str]): The CORS allow origins for the Chroma server.
- chroma_server_host (Optional[str]): The host for the Chroma server.
- chroma_server_http_port (Optional[int]): The HTTP port for the Chroma server.
- chroma_server_grpc_port (Optional[int]): The gRPC port for the Chroma server.
- allow_duplicates (bool): Whether to allow duplicates in the Vector Store.
Returns:
- Union[VectorStore, BaseRetriever]: The Vector Store or BaseRetriever object.
""" """
return self._build_chroma()
def _build_chroma(self) -> Chroma:
"""
Builds the Chroma object.
"""
# Chroma settings # Chroma settings
chroma_settings = None chroma_settings = None
client = None client = None
if chroma_server_host is not None: if self.chroma_server_host:
chroma_settings = Settings( chroma_settings = Settings(
chroma_server_cors_allow_origins=chroma_server_cors_allow_origins or [], chroma_server_cors_allow_origins=self.chroma_server_cors_allow_origins or [],
chroma_server_host=chroma_server_host, chroma_server_host=self.chroma_server_host,
chroma_server_http_port=chroma_server_http_port or None, chroma_server_http_port=self.chroma_server_http_port or None,
chroma_server_grpc_port=chroma_server_grpc_port or None, chroma_server_grpc_port=self.chroma_server_grpc_port or None,
chroma_server_ssl_enabled=chroma_server_ssl_enabled, chroma_server_ssl_enabled=self.chroma_server_ssl_enabled,
) )
client = chromadb.HttpClient(settings=chroma_settings) client = chromadb.Client(settings=chroma_settings)
# Check index_directory and expand it if it is a relative path # Check persist_directory and expand it if it is a relative path
if index_directory is not None: if self.persist_directory is not None:
index_directory = self.resolve_path(index_directory) persist_directory = self.resolve_path(self.persist_directory)
else:
persist_directory = None
chroma = Chroma( chroma = Chroma(
persist_directory=index_directory, persist_directory=persist_directory,
client=client, client=client,
embedding_function=embedding, embedding_function=self.embedding,
collection_name=collection_name, collection_name=self.collection_name,
) )
if allow_duplicates:
if self.add_to_vector_store:
self._add_documents_to_vector_store(chroma)
self.status = chroma_collection_to_data(chroma.get())
return chroma
def _add_documents_to_vector_store(self, chroma: Chroma) -> None:
"""
Adds documents to the Vector Store.
"""
if self.allow_duplicates:
stored_data = [] stored_data = []
else: else:
stored_data = chroma_collection_to_data(chroma.get()) stored_data = chroma_collection_to_data(chroma.get())
@ -120,16 +186,54 @@ class ChromaComponent(CustomComponent):
for value in deepcopy(stored_data): for value in deepcopy(stored_data):
del value.id del value.id
_stored_documents_without_id.append(value) _stored_documents_without_id.append(value)
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
if _input not in _stored_documents_without_id: if _input not in _stored_documents_without_id:
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
raise ValueError("Inputs must be a Data objects.") raise ValueError("Vector Store Inputs must be Data objects.")
if documents and embedding is not None: if documents and self.embedding is not None:
logger.debug(f"Adding {len(documents)} documents to the Vector Store.")
chroma.add_documents(documents) chroma.add_documents(documents)
else:
logger.debug("No documents to add to the Vector Store.")
self.status = stored_data def search_documents(self) -> List[Data]:
return chroma """
Search for documents in the Chroma vector store.
"""
if not self.search_input:
return
vector_store = self._build_chroma()
logger.debug(f"Search input: {self.search_input}")
logger.debug(f"Search type: {self.search_type}")
logger.debug(f"Number of results: {self.number_of_results}")
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
if self.search_type == "Similarity":
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
elif self.search_type == "MMR":
docs = vector_store.max_marginal_relevance_search(
query=self.search_input,
k=self.number_of_results,
)
else:
raise ValueError(f"Invalid search type: {self.search_type}")
logger.debug(f"Retrieved documents: {len(docs)}")
data = docs_to_data(docs)
logger.debug(f"Converted documents to data: {len(data)}")
self.status = data
return data
else:
logger.debug("No search input provided. Skipping search.")
return []

View file

@ -1,94 +1,119 @@
from datetime import timedelta from datetime import timedelta
from typing import List, Optional, Union from typing import List, Optional, Union
from langchain_core.retrievers import BaseRetriever from langchain_community.vectorstores import CouchbaseVectorStore
from langchain.embeddings.base import Embeddings
from langchain.schema import BaseRetriever
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings, VectorStore from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class CouchbaseVectorStoreComponent(Component):
class CouchbaseComponent(CustomComponent):
display_name = "Couchbase" display_name = "Couchbase"
description = "Construct a `Couchbase Vector Search` vector store from raw documents." description = "Couchbase Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/integrations/vectorstores/couchbase" documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/couchbase"
icon = "Couchbase" icon = "Couchbase"
field_order = [
"couchbase_connection_string", inputs = [
"couchbase_username", StrInput(name="couchbase_connection_string", display_name="Couchbase Cluster connection string", required=True),
"couchbase_password", StrInput(name="couchbase_username", display_name="Couchbase username", required=True),
"bucket_name", StrInput(name="couchbase_password", display_name="Couchbase password", password=True, required=True),
"scope_name", StrInput(name="bucket_name", display_name="Bucket Name", required=True),
"collection_name", StrInput(name="scope_name", display_name="Scope Name", required=True),
"index_name", StrInput(name="collection_name", display_name="Collection Name", required=True),
StrInput(name="index_name", display_name="Index Name", required=True),
HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
StrInput(name="search_input", display_name="Search Input"),
IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
] ]
def build_config(self): outputs = [
return { Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=CouchbaseVectorStore),
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]}, Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
"embedding": {"display_name": "Embedding"}, Output(display_name="Search Results", name="search_results", method="search_documents"),
"couchbase_connection_string": {"display_name": "Couchbase Cluster connection string", "required": True}, ]
"couchbase_username": {"display_name": "Couchbase username", "required": True},
"couchbase_password": {"display_name": "Couchbase password", "password": True, "required": True},
"bucket_name": {"display_name": "Bucket Name", "required": True},
"scope_name": {"display_name": "Scope Name", "required": True},
"collection_name": {"display_name": "Collection Name", "required": True},
"index_name": {"display_name": "Index Name", "required": True},
}
def build( def build_vector_store(self) -> CouchbaseVectorStore:
self, return self._build_couchbase()
embedding: Embeddings,
inputs: Optional[List[Data]] = None, def build_base_retriever(self) -> BaseRetriever:
bucket_name: str = "", return self._build_couchbase()
scope_name: str = "",
collection_name: str = "", def _build_couchbase(self) -> CouchbaseVectorStore:
index_name: str = "",
couchbase_connection_string: str = "",
couchbase_username: str = "",
couchbase_password: str = "",
) -> Union[VectorStore, BaseRetriever]:
try: try:
from couchbase.auth import PasswordAuthenticator # type: ignore from couchbase.auth import PasswordAuthenticator # type: ignore
from couchbase.cluster import Cluster # type: ignore from couchbase.cluster import Cluster # type: ignore
from couchbase.options import ClusterOptions # type: ignore from couchbase.options import ClusterOptions # type: ignore
from langchain_community.vectorstores import CouchbaseVectorStore
except ImportError as e: except ImportError as e:
raise ImportError( raise ImportError(
"Failed to import Couchbase dependencies. Install it using `pip install langflow[couchbase] --pre`" "Failed to import Couchbase dependencies. Install it using `pip install langflow[couchbase] --pre`"
) from e ) from e
try: try:
auth = PasswordAuthenticator(couchbase_username, couchbase_password) auth = PasswordAuthenticator(self.couchbase_username, self.couchbase_password)
options = ClusterOptions(auth) options = ClusterOptions(auth)
cluster = Cluster(couchbase_connection_string, options) cluster = Cluster(self.couchbase_connection_string, options)
cluster.wait_until_ready(timedelta(seconds=5)) cluster.wait_until_ready(timedelta(seconds=5))
except Exception as e: except Exception as e:
raise ValueError(f"Failed to connect to Couchbase: {e}") raise ValueError(f"Failed to connect to Couchbase: {e}")
if self.add_to_vector_store:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
if documents: if documents:
vector_store = CouchbaseVectorStore.from_documents( couchbase_vs = CouchbaseVectorStore.from_documents(
documents=documents, documents=documents,
cluster=cluster, cluster=cluster,
bucket_name=bucket_name, bucket_name=self.bucket_name,
scope_name=scope_name, scope_name=self.scope_name,
collection_name=collection_name, collection_name=self.collection_name,
embedding=embedding, embedding=self.embedding,
index_name=index_name, index_name=self.index_name,
) )
else: else:
vector_store = CouchbaseVectorStore( couchbase_vs = CouchbaseVectorStore(
cluster=cluster, cluster=cluster,
bucket_name=bucket_name, bucket_name=self.bucket_name,
scope_name=scope_name, scope_name=self.scope_name,
collection_name=collection_name, collection_name=self.collection_name,
embedding=embedding, embedding=self.embedding,
index_name=index_name, index_name=self.index_name,
) )
return vector_store else:
couchbase_vs = CouchbaseVectorStore(
cluster=cluster,
bucket_name=self.bucket_name,
scope_name=self.scope_name,
collection_name=self.collection_name,
embedding=self.embedding,
index_name=self.index_name,
)
return couchbase_vs
def search_documents(self) -> List[Data]:
vector_store = self._build_couchbase()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -1,46 +1,184 @@
from typing import List, Text, Union from copy import deepcopy
from typing import List, Optional, Union
from langchain_community.vectorstores.faiss import FAISS from langchain.vectorstores import FAISS
from langchain_core.retrievers import BaseRetriever from langchain.embeddings.base import Embeddings
from langchain_core.vectorstores import VectorStore from langchain.schema import BaseRetriever
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput
from langflow.schema import Data from langflow.schema import Data
from langflow.template import Output
from langflow.field_typing import Embeddings, Text
from langflow.helpers.data import docs_to_data
from loguru import logger
class FAISSComponent(CustomComponent): class FAISSVectorStoreComponent(Component):
display_name = "FAISS" """
description = "Ingest documents into FAISS Vector Store." FAISS Vector Store with search capabilities
"""
display_name: str = "FAISS"
description: str = "FAISS Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/faiss" documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/faiss"
icon = "FAISS"
def build_config(self): inputs = [
return { StrInput(
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]}, name="folder_path",
"embedding": {"display_name": "Embedding"}, display_name="Folder Path",
"folder_path": { info="Path to save the FAISS index. It will be relative to where Langflow is running.",
"display_name": "Folder Path", ),
"info": "Path to save the FAISS index. It will be relative to where Langflow is running.", StrInput(
}, name="index_name",
"index_name": {"display_name": "Index Name"}, display_name="Index Name",
} value="langflow_index",
),
HandleInput(
name="embedding", display_name="Embedding", input_types=["Embeddings"]
),
StrInput(
name="vector_store_inputs",
display_name="Vector Store Inputs",
input_types=["Document", "Data"],
is_list=True,
),
BoolInput(
name="add_to_vector_store",
display_name="Add to Vector Store",
info="If true, the Vector Store Inputs will be added to the Vector Store.",
),
BoolInput(
name="allow_dangerous_deserialization",
display_name="Allow Dangerous Deserialization",
info="Set to True to allow loading pickle files from untrusted sources. Only enable this if you trust the source of the data.",
advanced=True,
value=False,
),
StrInput(
name="search_input",
display_name="Search Input",
),
IntInput(
name="number_of_results",
display_name="Number of Results",
info="Number of results to return.",
advanced=True,
value=4,
),
]
def build( outputs = [
self, Output(
embedding: Embeddings, display_name="Vector Store",
inputs: List[Data], name="vector_store",
folder_path: str, method="build_vector_store",
index_name: str = "langflow_index", ),
) -> Union[VectorStore, FAISS, BaseRetriever]: Output(
display_name="Base Retriever",
name="base_retriever",
method="build_base_retriever",
),
Output(
display_name="Search Results",
name="search_results",
method="search_documents",
),
]
def build_vector_store(self) -> FAISS:
"""
Builds the Vector Store object.
"""
return self._build_faiss()
def build_base_retriever(self) -> BaseRetriever:
"""
Builds the BaseRetriever object.
"""
return self._build_faiss()
def _build_faiss(self) -> FAISS:
"""
Builds the FAISS object.
"""
if not self.folder_path:
raise ValueError("Folder path is required to save the FAISS index.")
path = self.resolve_path(self.folder_path)
if self.add_to_vector_store:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
vector_store = FAISS.from_documents(documents=documents, embedding=embedding)
if not folder_path: faiss = FAISS.from_documents(documents=documents, embedding=self.embedding)
raise ValueError("Folder path is required to save the FAISS index.") faiss.save_local(Text(path), self.index_name)
path = self.resolve_path(folder_path) else:
vector_store.save_local(Text(path), index_name) try:
return vector_store faiss = FAISS.load_local(
folder_path=Text(path),
embeddings=self.embedding,
index_name=self.index_name,
allow_dangerous_deserialization=self.allow_dangerous_deserialization,
)
except Exception as e:
raise ValueError(
"Failed to load the FAISS index. Make sure the index was created with trusted data. "
"If you trust the data source, you can set `allow_dangerous_deserialization` to `True` "
"in the component's advanced settings to enable deserialization."
) from e
return faiss
def search_documents(self) -> List[Data]:
"""
Search for documents in the FAISS vector store.
"""
if not self.folder_path:
raise ValueError("Folder path is required to load the FAISS index.")
path = self.resolve_path(self.folder_path)
try:
vector_store = FAISS.load_local(
folder_path=Text(path),
embeddings=self.embedding,
index_name=self.index_name,
allow_dangerous_deserialization=self.allow_dangerous_deserialization,
)
except Exception as e:
raise ValueError(
"Failed to load the FAISS index. Make sure the index was created with trusted data. "
"If you trust the data source, you can set `allow_dangerous_deserialization` to `True` "
"in the component's advanced settings to enable deserialization."
) from e
if not vector_store:
raise ValueError("Failed to load the FAISS index.")
logger.debug(f"Search input: {self.search_input}")
logger.debug(f"Number of results: {self.number_of_results}")
if (
self.search_input
and isinstance(self.search_input, str)
and self.search_input.strip()
):
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
logger.debug(f"Retrieved documents: {len(docs)}")
data = docs_to_data(docs)
logger.debug(f"Converted documents to data: {len(data)}")
logger.debug(data)
return data # Return the search results data
else:
logger.debug("No search input provided. Skipping search.")
return []

View file

@ -1,64 +1,101 @@
from typing import List, Optional from typing import List, Optional
from langchain_community.vectorstores.mongodb_atlas import MongoDBAtlasVectorSearch from langchain_community.vectorstores import MongoDBAtlasVectorSearch
from langchain.embeddings.base import Embeddings
from langchain.schema import BaseRetriever
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class MongoDBAtlasComponent(Component):
class MongoDBAtlasComponent(CustomComponent):
display_name = "MongoDB Atlas" display_name = "MongoDB Atlas"
description = "Construct a `MongoDB Atlas Vector Search` vector store from raw documents." description = "MongoDB Atlas Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/mongodb_atlas"
icon = "MongoDB" icon = "MongoDB"
def build_config(self): inputs = [
return { StrInput(name="mongodb_atlas_cluster_uri", display_name="MongoDB Atlas Cluster URI", required=True),
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]}, StrInput(name="db_name", display_name="Database Name", required=True),
"embedding": {"display_name": "Embedding"}, StrInput(name="collection_name", display_name="Collection Name", required=True),
"collection_name": {"display_name": "Collection Name"}, StrInput(name="index_name", display_name="Index Name", required=True),
"db_name": {"display_name": "Database Name"}, HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
"index_name": {"display_name": "Index Name"}, HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
"mongodb_atlas_cluster_uri": {"display_name": "MongoDB Atlas Cluster URI"}, BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
} StrInput(name="search_input", display_name="Search Input"),
IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
]
def build( outputs = [
self, Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=MongoDBAtlasVectorSearch),
embedding: Embeddings, Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
inputs: Optional[List[Data]] = None, Output(display_name="Search Results", name="search_results", method="search_documents"),
collection_name: str = "", ]
db_name: str = "",
index_name: str = "", def build_vector_store(self) -> MongoDBAtlasVectorSearch:
mongodb_atlas_cluster_uri: str = "", return self._build_mongodb_atlas()
) -> MongoDBAtlasVectorSearch:
def build_base_retriever(self) -> BaseRetriever:
return self._build_mongodb_atlas()
def _build_mongodb_atlas(self) -> MongoDBAtlasVectorSearch:
try: try:
from pymongo import MongoClient from pymongo import MongoClient
except ImportError: except ImportError:
raise ImportError("Please install pymongo to use MongoDB Atlas Vector Store") raise ImportError("Please install pymongo to use MongoDB Atlas Vector Store")
try: try:
mongo_client: MongoClient = MongoClient(mongodb_atlas_cluster_uri) mongo_client: MongoClient = MongoClient(self.mongodb_atlas_cluster_uri)
collection = mongo_client[db_name][collection_name] collection = mongo_client[self.db_name][self.collection_name]
except Exception as e: except Exception as e:
raise ValueError(f"Failed to connect to MongoDB Atlas: {e}") raise ValueError(f"Failed to connect to MongoDB Atlas: {e}")
if self.add_to_vector_store:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
if documents: if documents:
vector_store = MongoDBAtlasVectorSearch.from_documents( vector_store = MongoDBAtlasVectorSearch.from_documents(
documents=documents, documents=documents,
embedding=embedding, embedding=self.embedding,
collection=collection, collection=collection,
db_name=db_name, db_name=self.db_name,
index_name=index_name, index_name=self.index_name,
mongodb_atlas_cluster_uri=mongodb_atlas_cluster_uri, mongodb_atlas_cluster_uri=self.mongodb_atlas_cluster_uri,
) )
else: else:
vector_store = MongoDBAtlasVectorSearch( vector_store = MongoDBAtlasVectorSearch(
embedding=embedding, embedding=self.embedding,
collection=collection, collection=collection,
index_name=index_name, index_name=self.index_name,
) )
else:
vector_store = MongoDBAtlasVectorSearch(
embedding=self.embedding,
collection=collection,
index_name=self.index_name,
)
return vector_store return vector_store
def search_documents(self) -> List[Data]:
vector_store = self._build_mongodb_atlas()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -1,151 +1,87 @@
from typing import List, Optional, Union from typing import List, Optional, Union
from langchain_core.documents import Document from langchain.vectorstores import Pinecone
from langchain_core.retrievers import BaseRetriever from langchain.embeddings.base import Embeddings
from langchain_core.vectorstores import VectorStore from langchain.schema import BaseRetriever
from langchain_pinecone._utilities import DistanceStrategy
from langchain_pinecone.vectorstores import PineconeVectorStore
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput, DropdownInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class PineconeVectorStoreComponent(Component):
class PineconeComponent(CustomComponent):
display_name = "Pinecone" display_name = "Pinecone"
description = "Construct Pinecone wrapper from raw documents." description = "Pinecone Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/pinecone"
icon = "Pinecone" icon = "Pinecone"
field_order = ["index_name", "namespace", "distance_strategy", "pinecone_api_key", "documents", "embedding"]
def build_config(self): inputs = [
distance_options = [e.value.title().replace("_", " ") for e in DistanceStrategy] StrInput(name="index_name", display_name="Index Name", required=True),
distance_value = distance_options[0] StrInput(name="namespace", display_name="Namespace", info="Namespace for the index."),
return { DropdownInput(name="distance_strategy", display_name="Distance Strategy", options=["Cosine", "Euclidean", "Dot Product"], value="Cosine", advanced=True),
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]}, StrInput(name="pinecone_api_key", display_name="Pinecone API Key", password=True, required=True),
"embedding": {"display_name": "Embedding"}, HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
"index_name": {"display_name": "Index Name"}, StrInput(name="text_key", display_name="Text Key", info="Key in the record to use as text.", value="text", advanced=True),
"namespace": {"display_name": "Namespace"}, HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
"text_key": {"display_name": "Text Key"}, BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
"distance_strategy": { StrInput(name="search_input", display_name="Search Input"),
"display_name": "Distance Strategy", IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
# get values from enum ]
# and make them title case for display
"options": distance_options,
"advanced": True,
"value": distance_value,
},
"pinecone_api_key": {
"display_name": "Pinecone API Key",
"default": "",
"password": True,
"required": True,
},
"pool_threads": {
"display_name": "Pool Threads",
"default": 1,
"advanced": True,
},
}
def from_existing_index( outputs = [
self, Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=Pinecone),
index_name: str, Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
embedding: Embeddings, Output(display_name="Search Results", name="search_results", method="search_documents"),
pinecone_api_key: str | None, ]
text_key: str = "text",
namespace: Optional[str] = None,
distance_strategy: DistanceStrategy = DistanceStrategy.COSINE,
pool_threads: int = 4,
) -> PineconeVectorStore:
"""Load pinecone vectorstore from index name."""
pinecone_index = PineconeVectorStore.get_pinecone_index(
index_name, pool_threads, pinecone_api_key=pinecone_api_key
)
return PineconeVectorStore(
index=pinecone_index,
embedding=embedding,
text_key=text_key,
namespace=namespace,
distance_strategy=distance_strategy,
)
def from_documents( def build_vector_store(self) -> Pinecone:
self, return self._build_pinecone()
documents: List[Document],
embedding: Embeddings,
index_name: str,
pinecone_api_key: str | None,
text_key: str = "text",
namespace: Optional[str] = None,
pool_threads: int = 4,
distance_strategy: DistanceStrategy = DistanceStrategy.COSINE,
batch_size: int = 32,
upsert_kwargs: Optional[dict] = None,
embeddings_chunk_size: int = 1000,
) -> PineconeVectorStore:
"""Create a new pinecone vectorstore from documents."""
texts = [d.page_content for d in documents]
metadatas = [d.metadata for d in documents]
pinecone = self.from_existing_index(
index_name=index_name,
embedding=embedding,
pinecone_api_key=pinecone_api_key,
text_key=text_key,
namespace=namespace,
distance_strategy=distance_strategy,
pool_threads=pool_threads,
)
pinecone.add_texts(
texts,
metadatas=metadatas,
ids=None,
namespace=namespace,
batch_size=batch_size,
embedding_chunk_size=embeddings_chunk_size,
**(upsert_kwargs or {}),
)
return pinecone
def build( def build_base_retriever(self) -> BaseRetriever:
self, return self._build_pinecone()
embedding: Embeddings,
distance_strategy: str, def _build_pinecone(self) -> Pinecone:
inputs: Optional[List[Data]] = None, from langchain_pinecone._utilities import DistanceStrategy
text_key: str = "text", from langchain_pinecone.vectorstores import Pinecone
pool_threads: int = 4,
index_name: Optional[str] = None, distance_strategy = self.distance_strategy.replace(" ", "_").upper()
pinecone_api_key: Optional[str] = None,
namespace: Optional[str] = "default",
) -> Union[VectorStore, BaseRetriever]:
# get distance strategy from string
distance_strategy = distance_strategy.replace(" ", "_").upper()
_distance_strategy = DistanceStrategy[distance_strategy] _distance_strategy = DistanceStrategy[distance_strategy]
if not index_name:
raise ValueError("Index Name is required.") pinecone = Pinecone(
index_name=self.index_name,
embedding=self.embedding,
text_key=self.text_key,
namespace=self.namespace,
distance_strategy=_distance_strategy,
pinecone_api_key=self.pinecone_api_key,
)
if self.add_to_vector_store:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
if documents: if documents:
return self.from_documents( pinecone.add_documents(documents)
documents=documents,
embedding=embedding, return pinecone
index_name=index_name,
pinecone_api_key=pinecone_api_key, def search_documents(self) -> List[Data]:
text_key=text_key, vector_store = self._build_pinecone()
namespace=namespace,
distance_strategy=_distance_strategy, if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
pool_threads=pool_threads, docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
) )
return self.from_existing_index( data = docs_to_data(docs)
index_name=index_name, self.status = data
embedding=embedding, return data
pinecone_api_key=pinecone_api_key, else:
text_key=text_key, return []
namespace=namespace,
distance_strategy=_distance_strategy,
pool_threads=pool_threads,
)

View file

@ -1,114 +1,115 @@
from typing import Optional, Union from typing import List, Optional
from langchain_community.vectorstores.qdrant import Qdrant from langchain.vectorstores import Qdrant
from langchain_core.retrievers import BaseRetriever from langchain.embeddings.base import Embeddings
from langchain_core.vectorstores import VectorStore from langchain.schema import BaseRetriever
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput, DropdownInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class QdrantVectorStoreComponent(Component):
class QdrantComponent(CustomComponent):
display_name = "Qdrant" display_name = "Qdrant"
description = "Construct Qdrant wrapper from a list of texts." description = "Qdrant Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/qdrant"
icon = "Qdrant" icon = "Qdrant"
def build_config(self): inputs = [
return { StrInput(name="collection_name", display_name="Collection Name", required=True),
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]}, StrInput(name="host", display_name="Host", value="localhost", advanced=True),
"embedding": {"display_name": "Embedding"}, IntInput(name="port", display_name="Port", value=6333, advanced=True),
"api_key": {"display_name": "API Key", "password": True, "advanced": True}, IntInput(name="grpc_port", display_name="gRPC Port", value=6334, advanced=True),
"collection_name": {"display_name": "Collection Name"}, StrInput(name="api_key", display_name="API Key", password=True, advanced=True),
"content_payload_key": { StrInput(name="prefix", display_name="Prefix", advanced=True),
"display_name": "Content Payload Key", IntInput(name="timeout", display_name="Timeout", advanced=True),
"advanced": True, StrInput(name="path", display_name="Path", advanced=True),
}, StrInput(name="url", display_name="URL", advanced=True),
"distance_func": {"display_name": "Distance Function", "advanced": True}, DropdownInput(name="distance_func", display_name="Distance Function", options=["Cosine", "Euclidean", "Dot Product"], value="Cosine", advanced=True),
"grpc_port": {"display_name": "gRPC Port", "advanced": True}, StrInput(name="content_payload_key", display_name="Content Payload Key", value="page_content", advanced=True),
"host": {"display_name": "Host", "advanced": True}, StrInput(name="metadata_payload_key", display_name="Metadata Payload Key", value="metadata", advanced=True),
"https": {"display_name": "HTTPS", "advanced": True}, HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
"location": {"display_name": "Location", "advanced": True}, HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
"metadata_payload_key": { BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
"display_name": "Metadata Payload Key", StrInput(name="search_input", display_name="Search Input"),
"advanced": True, IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
}, ]
"path": {"display_name": "Path", "advanced": True},
"port": {"display_name": "Port", "advanced": True}, outputs = [
"prefer_grpc": {"display_name": "Prefer gRPC", "advanced": True}, Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=Qdrant),
"prefix": {"display_name": "Prefix", "advanced": True}, Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
"timeout": {"display_name": "Timeout", "advanced": True}, Output(display_name="Search Results", name="search_results", method="search_documents"),
"url": {"display_name": "URL", "advanced": True}, ]
def build_vector_store(self) -> Qdrant:
return self._build_qdrant()
def build_base_retriever(self) -> BaseRetriever:
return self._build_qdrant()
def _build_qdrant(self) -> Qdrant:
qdrant_kwargs = {
"collection_name": self.collection_name,
"content_payload_key": self.content_payload_key,
"distance_func": self.distance_func,
"metadata_payload_key": self.metadata_payload_key,
} }
def build( server_kwargs = {
self, "host": self.host,
embedding: Embeddings, "port": self.port,
collection_name: str, "grpc_port": self.grpc_port,
inputs: Optional[Data] = None, "api_key": self.api_key,
api_key: Optional[str] = None, "prefix": self.prefix,
content_payload_key: str = "page_content", "timeout": self.timeout,
distance_func: str = "Cosine", "path": self.path,
grpc_port: int = 6334, "url": self.url,
https: bool = False, }
host: Optional[str] = None,
location: Optional[str] = None, # Remove None values from server_kwargs
metadata_payload_key: str = "metadata", server_kwargs = {k: v for k, v in server_kwargs.items() if v is not None}
path: Optional[str] = None,
port: Optional[int] = 6333, if self.add_to_vector_store:
prefer_grpc: bool = False,
prefix: Optional[str] = None,
timeout: Optional[int] = None,
url: Optional[str] = None,
) -> Union[VectorStore, Qdrant, BaseRetriever]:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
if not documents:
if documents:
qdrant = Qdrant.from_documents(
documents,
embedding=self.embedding,
client_kwargs=server_kwargs,
**qdrant_kwargs
)
else:
from qdrant_client import QdrantClient from qdrant_client import QdrantClient
client = QdrantClient( client = QdrantClient(**server_kwargs)
location=location, qdrant = Qdrant(embedding_function=self.embedding.embed_query, client=client, **qdrant_kwargs)
url=url,
port=port,
grpc_port=grpc_port,
https=https,
prefix=prefix,
timeout=timeout,
prefer_grpc=prefer_grpc,
api_key=api_key,
host=host,
path=path,
)
vs = Qdrant(
client=client,
collection_name=collection_name,
embeddings=embedding,
content_payload_key=content_payload_key,
metadata_payload_key=metadata_payload_key,
)
return vs
else: else:
vs = Qdrant.from_documents( from qdrant_client import QdrantClient
documents=documents, # type: ignore
embedding=embedding, client = QdrantClient(**server_kwargs)
api_key=api_key, qdrant = Qdrant(embedding_function=self.embedding.embed_query, client=client, **qdrant_kwargs)
collection_name=collection_name,
content_payload_key=content_payload_key, return qdrant
distance_func=distance_func,
grpc_port=grpc_port, def search_documents(self) -> List[Data]:
host=host, vector_store = self._build_qdrant()
https=https,
location=location, if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
metadata_payload_key=metadata_payload_key, docs = vector_store.similarity_search(
path=path, query=self.search_input,
port=port, k=self.number_of_results,
prefer_grpc=prefer_grpc,
prefix=prefix,
timeout=timeout,
url=url,
) )
return vs
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -1,49 +1,85 @@
from typing import List, Optional, Union from typing import List, Optional, Union
from langchain_community.vectorstores.supabase import SupabaseVectorStore from langchain_community.vectorstores import SupabaseVectorStore
from langchain_core.retrievers import BaseRetriever from langchain.embeddings.base import Embeddings
from langchain_core.vectorstores import VectorStore from langchain.schema import BaseRetriever
from supabase.client import Client, create_client from supabase.client import Client, create_client
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import IntInput, StrInput, HandleInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class SupabaseVectorStoreComponent(Component):
class SupabaseComponent(CustomComponent):
display_name = "Supabase" display_name = "Supabase"
description = "Return VectorStore initialized from texts and embeddings." description = "Supabase Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/supabase"
icon = "Supabase"
def build_config(self): inputs = [
return { StrInput(name="supabase_url", display_name="Supabase URL", required=True),
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]}, StrInput(name="supabase_service_key", display_name="Supabase Service Key", required=True),
"embedding": {"display_name": "Embedding"}, StrInput(name="table_name", display_name="Table Name", advanced=True),
"query_name": {"display_name": "Query Name"}, StrInput(name="query_name", display_name="Query Name"),
"supabase_service_key": {"display_name": "Supabase Service Key"}, HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
"supabase_url": {"display_name": "Supabase URL"}, HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
"table_name": {"display_name": "Table Name", "advanced": True}, StrInput(name="search_input", display_name="Search Input"),
} IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
]
outputs = [
Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=SupabaseVectorStore),
Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
Output(display_name="Search Results", name="search_results", method="search_documents"),
]
def build_vector_store(self) -> SupabaseVectorStore:
return self._build_supabase()
def build_base_retriever(self) -> BaseRetriever:
return self._build_supabase()
def _build_supabase(self) -> SupabaseVectorStore:
supabase: Client = create_client(self.supabase_url, supabase_key=self.supabase_service_key)
def build(
self,
embedding: Embeddings,
inputs: Optional[List[Data]] = None,
query_name: str = "",
supabase_service_key: str = "",
supabase_url: str = "",
table_name: str = "",
) -> Union[VectorStore, SupabaseVectorStore, BaseRetriever]:
supabase: Client = create_client(supabase_url, supabase_key=supabase_service_key)
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
return SupabaseVectorStore.from_documents(
if documents:
supabase_vs = SupabaseVectorStore.from_documents(
documents=documents, documents=documents,
embedding=embedding, embedding=self.embedding,
query_name=query_name, query_name=self.query_name,
client=supabase, client=supabase,
table_name=table_name, table_name=self.table_name,
) )
else:
supabase_vs = SupabaseVectorStore(
client=supabase,
embedding=self.embedding,
table_name=self.table_name,
query_name=self.query_name,
)
return supabase_vs
def search_documents(self) -> List[Data]:
vector_store = self._build_supabase()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -1,89 +1,101 @@
from typing import List, Optional, Union from typing import List, Optional, Union
from langchain_community.vectorstores.upstash import UpstashVectorStore from langchain_community.vectorstores import UpstashVectorStore
from langchain_core.embeddings import Embeddings from langchain.embeddings.base import Embeddings
from langchain_core.retrievers import BaseRetriever from langchain.schema import BaseRetriever
from langchain_core.vectorstores import VectorStore
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class UpstashVectorStoreComponent(Component):
display_name = "Upstash"
description = "Upstash Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/upstash"
icon = "Upstash"
class UpstashVectorStoreComponent(CustomComponent): inputs = [
""" StrInput(name="index_url", display_name="Index URL", info="The URL of the Upstash index.", required=True),
A custom component for implementing a Vector Store using Upstash. StrInput(name="index_token", display_name="Index Token", info="The token for the Upstash index.", required=True),
""" StrInput(name="text_key", display_name="Text Key", info="The key in the record to use as text.", value="text", advanced=True),
HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"], info="To use Upstash's embeddings, don't provide an embedding."),
HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
StrInput(name="search_input", display_name="Search Input"),
IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
]
display_name: str = "Upstash" outputs = [
description: str = "Create and Utilize an Upstash Vector Store" Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=UpstashVectorStore),
Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
Output(display_name="Search Results", name="search_results", method="search_documents"),
]
def build_config(self): def build_vector_store(self) -> UpstashVectorStore:
""" return self._build_upstash()
Builds the configuration for the component.
Returns: def build_base_retriever(self) -> BaseRetriever:
- dict: A dictionary containing the configuration options for the component. return self._build_upstash()
"""
return {
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]},
"embedding": {
"display_name": "Embedding",
"input_types": ["Embeddings"],
"info": "To use Upstash's embeddings, don't provide an embedding.",
},
"index_url": {
"display_name": "Index URL",
"info": "The URL of the Upstash index.",
},
"index_token": {
"display_name": "Index Token",
"info": "The token for the Upstash index.",
},
"text_key": {
"display_name": "Text Key",
"info": "The key in the record to use as text.",
"advanced": True,
},
}
def build( def _build_upstash(self) -> UpstashVectorStore:
self, use_upstash_embedding = self.embedding is None
inputs: Optional[List[Data]] = None,
text_key: str = "text", if self.add_to_vector_store:
index_url: Optional[str] = None,
index_token: Optional[str] = None,
embedding: Optional[Embeddings] = None,
) -> Union[VectorStore, BaseRetriever]:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
use_upstash_embedding = embedding is None if documents:
if not documents:
upstash_vs = UpstashVectorStore(
embedding=embedding or use_upstash_embedding,
text_key=text_key,
index_url=index_url,
index_token=index_token,
)
else:
if use_upstash_embedding: if use_upstash_embedding:
upstash_vs = UpstashVectorStore( upstash_vs = UpstashVectorStore(
embedding=use_upstash_embedding, embedding=use_upstash_embedding,
text_key=text_key, text_key=self.text_key,
index_url=index_url, index_url=self.index_url,
index_token=index_token, index_token=self.index_token,
) )
upstash_vs.add_documents(documents) upstash_vs.add_documents(documents)
elif embedding: else:
upstash_vs = UpstashVectorStore.from_documents( upstash_vs = UpstashVectorStore.from_documents(
documents=documents, # type: ignore documents=documents,
embedding=embedding, embedding=self.embedding,
text_key=text_key, text_key=self.text_key,
index_url=index_url, index_url=self.index_url,
index_token=index_token, index_token=self.index_token,
) )
else:
upstash_vs = UpstashVectorStore(
embedding=self.embedding or use_upstash_embedding,
text_key=self.text_key,
index_url=self.index_url,
index_token=self.index_token,
)
else:
upstash_vs = UpstashVectorStore(
embedding=self.embedding or use_upstash_embedding,
text_key=self.text_key,
index_url=self.index_url,
index_token=self.index_token,
)
return upstash_vs return upstash_vs
def search_documents(self) -> List[Data]:
vector_store = self._build_upstash()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -1,56 +1,53 @@
import tempfile import tempfile
import urllib import urllib
import urllib.request import urllib.request
from typing import List, Optional, Union from typing import List, Optional
from langchain_community.embeddings import FakeEmbeddings from langchain.embeddings import FakeEmbeddings
from langchain_community.vectorstores.vectara import Vectara from langchain.vectorstores import Vectara
from langchain_core.vectorstores import VectorStore from langchain.schema import BaseRetriever
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import BaseRetriever from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class VectaraVectorStoreComponent(Component):
class VectaraComponent(CustomComponent): display_name = "Vectara"
display_name: str = "Vectara" description = "Vectara Vector Store with search capabilities"
description: str = "Implementation of Vector Store using Vectara" documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/vectara"
documentation = "https://python.langchain.com/docs/integrations/vectorstores/vectara"
icon = "Vectara" icon = "Vectara"
field_config = {
"vectara_customer_id": {
"display_name": "Vectara Customer ID",
},
"vectara_corpus_id": {
"display_name": "Vectara Corpus ID",
},
"vectara_api_key": {
"display_name": "Vectara API Key",
"password": True,
},
"inputs": {
"display_name": "Input",
"input_types": ["Document", "Data"],
"info": "If provided, will be upserted to corpus (optional)",
},
"files_url": {
"display_name": "Files Url",
"info": "Make vectara object using url of files (optional)",
},
}
def build( inputs = [
self, StrInput(name="vectara_customer_id", display_name="Vectara Customer ID", required=True),
vectara_customer_id: str, StrInput(name="vectara_corpus_id", display_name="Vectara Corpus ID", required=True),
vectara_corpus_id: str, StrInput(name="vectara_api_key", display_name="Vectara API Key", password=True, required=True),
vectara_api_key: str, HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
files_url: Optional[List[str]] = None, BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
inputs: Optional[Data] = None, StrInput(name="search_input", display_name="Search Input"),
) -> Union[VectorStore, BaseRetriever]: IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
]
outputs = [
Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=Vectara),
Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
Output(display_name="Search Results", name="search_results", method="search_documents"),
]
def build_vector_store(self) -> Vectara:
return self._build_vectara()
def build_base_retriever(self) -> BaseRetriever:
return self._build_vectara()
def _build_vectara(self) -> Vectara:
source = "Langflow" source = "Langflow"
if self.add_to_vector_store:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
@ -58,33 +55,32 @@ class VectaraComponent(CustomComponent):
if documents: if documents:
return Vectara.from_documents( return Vectara.from_documents(
documents=documents, # type: ignore documents=documents,
embedding=FakeEmbeddings(size=768), embedding=FakeEmbeddings(size=768),
vectara_customer_id=vectara_customer_id, customer_id=self.vectara_customer_id,
vectara_corpus_id=vectara_corpus_id, corpus_id=self.vectara_corpus_id,
vectara_api_key=vectara_api_key, api_key=self.vectara_api_key,
source=source,
)
if files_url is not None:
files_list = []
for url in files_url:
name = tempfile.NamedTemporaryFile().name
urllib.request.urlretrieve(url, name)
files_list.append(name)
return Vectara.from_files(
files=files_list,
embedding=FakeEmbeddings(size=768),
vectara_customer_id=vectara_customer_id,
vectara_corpus_id=vectara_corpus_id,
vectara_api_key=vectara_api_key,
source=source, source=source,
) )
return Vectara( return Vectara(
vectara_customer_id=vectara_customer_id, customer_id=self.vectara_customer_id,
vectara_corpus_id=vectara_corpus_id, corpus_id=self.vectara_corpus_id,
vectara_api_key=vectara_api_key, api_key=self.vectara_api_key,
source=source, source=source,
) )
def search_documents(self) -> List[Data]:
vector_store = self._build_vectara()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -1,108 +1,91 @@
from typing import Optional, Union from typing import List, Optional, Union
import weaviate # type: ignore import weaviate
from langchain_community.vectorstores import Weaviate from langchain.vectorstores import Weaviate
from langchain_core.documents import Document from langchain.embeddings.base import Embeddings
from langchain_core.embeddings import Embeddings from langchain.schema import BaseRetriever
from langchain_core.retrievers import BaseRetriever
from langchain_core.vectorstores import VectorStore
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class WeaviateVectorStoreComponent(Component):
display_name = "Weaviate"
description = "Weaviate Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/weaviate"
icon = "Weaviate"
class WeaviateVectorStoreComponent(CustomComponent): inputs = [
display_name: str = "Weaviate" StrInput(name="url", display_name="Weaviate URL", value="http://localhost:8080", required=True),
description: str = "Implementation of Vector Store using Weaviate" StrInput(name="api_key", display_name="API Key", password=True, required=False),
documentation = "https://python.langchain.com/docs/integrations/vectorstores/weaviate" StrInput(name="index_name", display_name="Index Name", required=True),
field_config = { StrInput(name="text_key", display_name="Text Key", value="text", advanced=True),
"url": {"display_name": "Weaviate URL", "value": "http://localhost:8080"}, HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
"api_key": { HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
"display_name": "API Key", BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
"password": True, StrInput(name="search_input", display_name="Search Input"),
"required": False, IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
}, BoolInput(name="search_by_text", display_name="Search By Text", advanced=True),
"index_name": { ]
"display_name": "Index name",
"required": False,
},
"text_key": {
"display_name": "Text Key",
"required": False,
"advanced": True,
"value": "text",
},
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]},
"embedding": {"display_name": "Embedding"},
"attributes": {
"display_name": "Attributes",
"required": False,
"is_list": True,
"field_type": "str",
"advanced": True,
},
"search_by_text": {
"display_name": "Search By Text",
"field_type": "bool",
"advanced": True,
},
"code": {"show": False},
}
def build( outputs = [
self, Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=Weaviate),
url: str, Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
index_name: str, Output(display_name="Search Results", name="search_results", method="search_documents"),
search_by_text: bool = False, ]
api_key: Optional[str] = None,
text_key: str = "text", def build_vector_store(self) -> Weaviate:
embedding: Optional[Embeddings] = None, return self._build_weaviate()
inputs: Optional[Data] = None,
attributes: Optional[list] = None, def build_base_retriever(self) -> BaseRetriever:
) -> Union[VectorStore, BaseRetriever]: return self._build_weaviate()
if api_key:
auth_config = weaviate.AuthApiKey(api_key=api_key) def _build_weaviate(self) -> Weaviate:
client = weaviate.Client(url=url, auth_client_secret=auth_config) if self.api_key:
auth_config = weaviate.AuthApiKey(api_key=self.api_key)
client = weaviate.Client(url=self.url, auth_client_secret=auth_config)
else: else:
client = weaviate.Client(url=url) client = weaviate.Client(url=self.url)
def _to_pascal_case(word: str): if self.add_to_vector_store:
if word and not word[0].isupper(): documents = []
word = word.capitalize() for _input in self.vector_store_inputs or []:
if word.isidentifier():
return word
word = word.replace("-", " ").replace("_", " ")
parts = word.split()
pascal_case_word = "".join([part.capitalize() for part in parts])
return pascal_case_word
index_name = _to_pascal_case(index_name) if index_name else None
if not index_name:
raise ValueError("Index name is required")
documents: list[Document] = []
for _input in inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
elif isinstance(_input, Document): else:
documents.append(_input) documents.append(_input)
if documents and embedding is not None: if documents and self.embedding:
return Weaviate.from_documents( return Weaviate.from_documents(
client=client, client=client,
index_name=index_name, index_name=self.index_name,
documents=documents, documents=documents,
embedding=embedding, embedding=self.embedding,
by_text=search_by_text, by_text=self.search_by_text,
) )
return Weaviate( return Weaviate(
client=client, client=client,
index_name=index_name, index_name=self.index_name,
text_key=text_key, text_key=self.text_key,
embedding=embedding, embedding=self.embedding,
by_text=search_by_text, by_text=self.search_by_text,
attributes=attributes if attributes is not None else [],
) )
def search_documents(self) -> List[Data]:
vector_store = self._build_weaviate()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
data = docs_to_data(docs)
self.status = data
return data
else:
return []

View file

@ -1,81 +1,86 @@
from typing import Optional, Union from typing import List, Optional, Union
from langchain_community.vectorstores.pgvector import PGVector from langchain_community.vectorstores import PGVector
from langchain_core.embeddings import Embeddings from langchain.embeddings.base import Embeddings
from langchain_core.retrievers import BaseRetriever from langchain.schema import BaseRetriever
from langchain_core.vectorstores import VectorStore
from langflow.custom import CustomComponent from langflow.custom import Component
from langflow.field_typing import Embeddings, Text
from langflow.schema import Data from langflow.schema import Data
from langflow.inputs import BoolInput, IntInput, StrInput, HandleInput
from langflow.template import Output
from langflow.helpers.data import docs_to_data
class PGVectorComponent(Component):
display_name = "PGVector"
description = "PGVector Vector Store with search capabilities"
documentation = "https://python.langchain.com/docs/modules/data_connection/vectorstores/integrations/pgvector"
icon = "PGVector"
class PGVectorComponent(CustomComponent): inputs = [
""" StrInput(name="pg_server_url", display_name="PostgreSQL Server Connection String", required=True),
A custom component for implementing a Vector Store using PostgreSQL. StrInput(name="collection_name", display_name="Table", required=True),
""" HandleInput(name="embedding", display_name="Embedding", input_types=["Embeddings"]),
HandleInput(name="vector_store_inputs", display_name="Vector Store Inputs", input_types=["Document", "Data"], is_list=True),
BoolInput(name="add_to_vector_store", display_name="Add to Vector Store", info="If true, the Vector Store Inputs will be added to the Vector Store."),
StrInput(name="search_input", display_name="Search Input"),
IntInput(name="number_of_results", display_name="Number of Results", info="Number of results to return.", value=4, advanced=True),
]
display_name: str = "PGVector" outputs = [
description: str = "Implementation of Vector Store using PostgreSQL" Output(display_name="Vector Store", name="vector_store", method="build_vector_store", output_type=PGVector),
documentation = "https://python.langchain.com/docs/integrations/vectorstores/pgvector" Output(display_name="Base Retriever", name="base_retriever", method="build_base_retriever", output_type=BaseRetriever),
Output(display_name="Search Results", name="search_results", method="search_documents"),
]
def build_config(self): def build_vector_store(self) -> PGVector:
""" return self._build_pgvector()
Builds the configuration for the component.
Returns: def build_base_retriever(self) -> BaseRetriever:
- dict: A dictionary containing the configuration options for the component. return self._build_pgvector()
"""
return {
"code": {"show": False},
"inputs": {"display_name": "Input", "input_types": ["Document", "Data"]},
"embedding": {"display_name": "Embedding"},
"pg_server_url": {
"display_name": "PostgreSQL Server Connection String",
"advanced": False,
},
"collection_name": {"display_name": "Table", "advanced": False},
}
def build(
self,
embedding: Embeddings,
pg_server_url: str,
collection_name: str,
inputs: Optional[Data] = None,
) -> Union[VectorStore, BaseRetriever]:
"""
Builds the Vector Store or BaseRetriever object.
Args:
- embedding (Embeddings): The embeddings to use for the Vector Store.
- documents (Optional[Document]): The documents to use for the Vector Store.
- collection_name (str): The name of the PG table.
- pg_server_url (str): The URL for the PG server.
Returns:
- VectorStore: The Vector Store object.
"""
def _build_pgvector(self) -> PGVector:
if self.add_to_vector_store:
documents = [] documents = []
for _input in inputs or []: for _input in self.vector_store_inputs or []:
if isinstance(_input, Data): if isinstance(_input, Data):
documents.append(_input.to_lc_document()) documents.append(_input.to_lc_document())
else: else:
documents.append(_input) documents.append(_input)
try:
if documents is None: if documents:
vector_store = PGVector.from_existing_index( pgvector = PGVector.from_documents(
embedding=embedding, embedding=self.embedding,
collection_name=collection_name, documents=documents,
connection_string=pg_server_url, collection_name=self.collection_name,
connection_string=self.pg_server_url,
) )
else: else:
vector_store = PGVector.from_documents( pgvector = PGVector.from_existing_index(
embedding=embedding, embedding=self.embedding,
documents=documents, # type: ignore collection_name=self.collection_name,
collection_name=collection_name, connection_string=self.pg_server_url,
connection_string=pg_server_url,
) )
except Exception as e: else:
raise RuntimeError(f"Failed to build PGVector: {e}") pgvector = PGVector.from_existing_index(
return vector_store embedding=self.embedding,
collection_name=self.collection_name,
connection_string=self.pg_server_url,
)
return pgvector
def search_documents(self) -> List[Data]:
vector_store = self._build_pgvector()
if self.search_input and isinstance(self.search_input, str) and self.search_input.strip():
docs = vector_store.similarity_search(
query=self.search_input,
k=self.number_of_results,
)
data = docs_to_data(docs)
self.status = data
return data
else:
return []