feat: add StructuredOutput component (#4024)
* Add utility functions to build Pydantic models from schema definitions * Add unit tests for build_model_from_schema function in test_base_model.py - Implement various test cases to validate the functionality of build_model_from_schema. - Test cases cover scenarios such as handling valid and empty schemas, managing unknown field types, and processing schemas with missing optional keys. - Ensure proper handling of nested list and dict types, and verify the function's efficiency with large schemas. - Confirm that the function raises exceptions for invalid input and handles duplicate field names correctly. * Refactor tests in `test_base_model.py` to improve type handling and error checking * Refactor output schema handling to use TableInput and build_model_from_schema * Update OpenAI model components and hierarchical crew setup - Refactor `OpenAIModelComponent` to use `TableInput` for `output_schema` and integrate `build_model_from_schema`. - Modify `HierarchicalCrewComponent` to use unpacking for base inputs. - Ensure consistent import statements across JSON files. - Improve error handling and logging for vector store operations. * Add chat result model with message building and execution logic - Implement `build_messages_and_runnable` to construct message lists and configure runnable models. - Add `get_chat_result` to execute language models with input messages, supporting streaming and custom configurations. - Handle exceptions with optional custom error messages. * Add "table" to DIRECT_TYPES in constants.py * Add support for DataFrame input validation in TableInput class * Add StructuredOutputComponent for generating structured outputs from language models * Enhance structured output component with improved input descriptions and schema naming * Convert DataFrame to list of dictionaries in TableInput validation * Remove pandas dependency and refactor schema handling in structured_output.py * Remove 'default' field from structured output schema and update field initialization * Add 'number' and 'text' types to type mapping and remove default value from field creation * Enhance error handling in structured output building process * Improve error message for non-BaseModel output in structured_output.py * Add unit tests for StructuredOutputComponent in helpers module - Implement various test cases to ensure correct functionality of StructuredOutputComponent. - Test successful structured output generation, handling of unsupported language models, and correct output model building. - Validate handling of multiple outputs, empty and invalid output schemas, and nested schemas. - Include tests for large input values and invalid language model configurations. * Update description for StructuredOutputComponent to clarify functionality * Add default values and error handling for structured output in helpers * Remove unused 'method' parameter from 'with_structured_output' in MockLanguageModel * refactor: rename test_base_model.py to test_base_model_from_schema.py Rename the test_base_model.py file to test_base_model_from_schema.py to better reflect its purpose of testing the build_model_from_schema function. This change improves code clarity and maintainability. * Add type ignore comments to suppress type checking errors * Add Generic typing to StructuredOutputComponent and fix method call * Revert "Refactor output schema handling to use TableInput and build_model_from_schema" This reverts commit 2e84a8608689bcfb519dc589d3eeef852784f3e4. * Deprecate JSON mode in OpenAIModel output schema documentation * Remove unused Generic import and add type ignore comment in StructuredOutputComponent * Refactor OpenAI model components and deprecate output schema - Refactored `OpenAIModelComponent` to use `operator.ior` and `functools.reduce` for converting `output_schema` to a dictionary. - Deprecated the `output_schema` field, updating its info to reflect the deprecation. - Simplified the `_docs_to_data` method in `SplitTextComponent` for better readability. - Updated import statements and removed unused imports across multiple JSON files. * Add specific type ignore comments and update exception types in backend code
This commit is contained in:
parent
c9b9fcf63c
commit
2be7c56939
19 changed files with 693 additions and 57 deletions
|
|
@ -0,0 +1,240 @@
|
|||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel
|
||||
|
||||
from langflow.components.helpers.structured_output import StructuredOutputComponent
|
||||
from langflow.schema.data import Data
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client():
|
||||
pass
|
||||
|
||||
|
||||
class TestStructuredOutputComponent:
|
||||
# Ensure that the structured output is successfully generated with the correct BaseModel instance returned by the mock function
|
||||
def test_successful_structured_output_generation_with_patch_with_config(self):
|
||||
from unittest.mock import patch
|
||||
|
||||
class MockLanguageModel:
|
||||
def with_structured_output(self, schema):
|
||||
return self
|
||||
|
||||
def with_config(self, config):
|
||||
return self
|
||||
|
||||
def invoke(self, inputs):
|
||||
return self
|
||||
|
||||
def mock_get_chat_result(runnable, input_value, config):
|
||||
class MockBaseModel(BaseModel):
|
||||
def model_dump(self):
|
||||
return {"field": "value"}
|
||||
|
||||
return MockBaseModel()
|
||||
|
||||
component = StructuredOutputComponent(
|
||||
llm=MockLanguageModel(),
|
||||
input_value="Test input",
|
||||
schema_name="TestSchema",
|
||||
output_schema=[{"name": "field", "type": "str", "description": "A test field"}],
|
||||
multiple=False,
|
||||
)
|
||||
|
||||
with patch("langflow.components.helpers.structured_output.get_chat_result", mock_get_chat_result):
|
||||
result = component.build_structured_output()
|
||||
assert isinstance(result, Data)
|
||||
assert result.data == {"field": "value"}
|
||||
|
||||
# Raises ValueError when the language model does not support structured output
|
||||
def test_raises_value_error_for_unsupported_language_model(self):
|
||||
# Mocking an incompatible language model
|
||||
class MockLanguageModel:
|
||||
pass
|
||||
|
||||
# Creating an instance of StructuredOutputComponent
|
||||
component = StructuredOutputComponent(
|
||||
llm=MockLanguageModel(),
|
||||
input_value="Test input",
|
||||
schema_name="TestSchema",
|
||||
output_schema=[{"name": "field", "type": "str", "description": "A test field"}],
|
||||
multiple=False,
|
||||
)
|
||||
|
||||
with pytest.raises(TypeError, match="Language model does not support structured output."):
|
||||
component.build_structured_output()
|
||||
|
||||
# Correctly builds the output model from the provided schema
|
||||
def test_correctly_builds_output_model(self):
|
||||
# Import internal organization modules, packages, and libraries
|
||||
from langflow.helpers.base_model import build_model_from_schema
|
||||
from langflow.inputs.inputs import TableInput
|
||||
|
||||
# Setup
|
||||
component = StructuredOutputComponent()
|
||||
schema = [
|
||||
{
|
||||
"name": "name",
|
||||
"display_name": "Name",
|
||||
"type": "str",
|
||||
"description": "Specify the name of the output field.",
|
||||
},
|
||||
{
|
||||
"name": "description",
|
||||
"display_name": "Description",
|
||||
"type": "str",
|
||||
"description": "Describe the purpose of the output field.",
|
||||
},
|
||||
{
|
||||
"name": "type",
|
||||
"display_name": "Type",
|
||||
"type": "str",
|
||||
"description": (
|
||||
"Indicate the data type of the output field " "(e.g., str, int, float, bool, list, dict)."
|
||||
),
|
||||
},
|
||||
{
|
||||
"name": "multiple",
|
||||
"display_name": "Multiple",
|
||||
"type": "boolean",
|
||||
"description": "Set to True if this output field should be a list of the specified type.",
|
||||
},
|
||||
]
|
||||
component.output_schema = TableInput(name="output_schema", display_name="Output Schema", table_schema=schema)
|
||||
|
||||
# Assertion
|
||||
output_model = build_model_from_schema(schema)
|
||||
assert isinstance(output_model, type)
|
||||
|
||||
# Properly handles multiple outputs when 'multiple' is set to True
|
||||
def test_handles_multiple_outputs(self):
|
||||
# Import internal organization modules, packages, and libraries
|
||||
from langflow.helpers.base_model import build_model_from_schema
|
||||
from langflow.inputs.inputs import TableInput
|
||||
|
||||
# Setup
|
||||
component = StructuredOutputComponent()
|
||||
schema = [
|
||||
{
|
||||
"name": "name",
|
||||
"display_name": "Name",
|
||||
"type": "str",
|
||||
"description": "Specify the name of the output field.",
|
||||
},
|
||||
{
|
||||
"name": "description",
|
||||
"display_name": "Description",
|
||||
"type": "str",
|
||||
"description": "Describe the purpose of the output field.",
|
||||
},
|
||||
{
|
||||
"name": "type",
|
||||
"display_name": "Type",
|
||||
"type": "str",
|
||||
"description": (
|
||||
"Indicate the data type of the output field " "(e.g., str, int, float, bool, list, dict)."
|
||||
),
|
||||
},
|
||||
{
|
||||
"name": "multiple",
|
||||
"display_name": "Multiple",
|
||||
"type": "boolean",
|
||||
"description": "Set to True if this output field should be a list of the specified type.",
|
||||
},
|
||||
]
|
||||
component.output_schema = TableInput(name="output_schema", display_name="Output Schema", table_schema=schema)
|
||||
component.multiple = True
|
||||
|
||||
# Assertion
|
||||
output_model = build_model_from_schema(schema)
|
||||
assert isinstance(output_model, type)
|
||||
|
||||
def test_empty_output_schema(self):
|
||||
component = StructuredOutputComponent(
|
||||
llm=MagicMock(),
|
||||
input_value="Test input",
|
||||
schema_name="EmptySchema",
|
||||
output_schema=[],
|
||||
multiple=False,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="Output schema cannot be empty"):
|
||||
component.build_structured_output()
|
||||
|
||||
def test_invalid_output_schema_type(self):
|
||||
component = StructuredOutputComponent(
|
||||
llm=MagicMock(),
|
||||
input_value="Test input",
|
||||
schema_name="InvalidSchema",
|
||||
output_schema=[{"name": "field", "type": "invalid_type", "description": "Invalid field"}],
|
||||
multiple=False,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="Invalid type: invalid_type"):
|
||||
component.build_structured_output()
|
||||
|
||||
@patch("langflow.components.helpers.structured_output.get_chat_result")
|
||||
def test_nested_output_schema(self, mock_get_chat_result):
|
||||
class ChildModel(BaseModel):
|
||||
child: str = "value"
|
||||
|
||||
class ParentModel(BaseModel):
|
||||
parent: ChildModel = ChildModel()
|
||||
|
||||
mock_llm = MagicMock()
|
||||
mock_llm.with_structured_output.return_value = mock_llm
|
||||
mock_get_chat_result.return_value = ParentModel(parent=ChildModel(child="value"))
|
||||
|
||||
component = StructuredOutputComponent(
|
||||
llm=mock_llm,
|
||||
input_value="Test input",
|
||||
schema_name="NestedSchema",
|
||||
output_schema=[
|
||||
{
|
||||
"name": "parent",
|
||||
"type": "dict",
|
||||
"description": "Parent field",
|
||||
"fields": [{"name": "child", "type": "str", "description": "Child field"}],
|
||||
}
|
||||
],
|
||||
multiple=False,
|
||||
)
|
||||
|
||||
result = component.build_structured_output()
|
||||
assert isinstance(result, Data)
|
||||
assert result.data == {"parent": {"child": "value"}}
|
||||
|
||||
@patch("langflow.components.helpers.structured_output.get_chat_result")
|
||||
def test_large_input_value(self, mock_get_chat_result):
|
||||
large_input = "Test input " * 1000
|
||||
|
||||
class MockBaseModel(BaseModel):
|
||||
field: str = "value"
|
||||
|
||||
mock_get_chat_result.return_value = MockBaseModel(field="value")
|
||||
|
||||
component = StructuredOutputComponent(
|
||||
llm=MagicMock(),
|
||||
input_value=large_input,
|
||||
schema_name="LargeInputSchema",
|
||||
output_schema=[{"name": "field", "type": "str", "description": "A test field"}],
|
||||
multiple=False,
|
||||
)
|
||||
|
||||
result = component.build_structured_output()
|
||||
assert isinstance(result, Data)
|
||||
assert result.data == {"field": "value"}
|
||||
mock_get_chat_result.assert_called_once()
|
||||
|
||||
def test_invalid_llm_config(self):
|
||||
component = StructuredOutputComponent(
|
||||
llm="invalid_llm", # Not a proper LLM instance
|
||||
input_value="Test input",
|
||||
schema_name="InvalidLLMSchema",
|
||||
output_schema=[{"name": "field", "type": "str", "description": "A test field"}],
|
||||
multiple=False,
|
||||
)
|
||||
|
||||
with pytest.raises(TypeError, match="Language model does not support structured output."):
|
||||
component.build_structured_output()
|
||||
160
src/backend/tests/unit/helpers/test_base_model_from_schema.py
Normal file
160
src/backend/tests/unit/helpers/test_base_model_from_schema.py
Normal file
|
|
@ -0,0 +1,160 @@
|
|||
# Generated by qodo Gen
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel
|
||||
from pydantic_core import PydanticUndefined
|
||||
|
||||
from langflow.helpers.base_model import build_model_from_schema
|
||||
|
||||
|
||||
class TestBuildModelFromSchema:
|
||||
# Successfully creates a Pydantic model from a valid schema
|
||||
def test_create_model_from_valid_schema(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value", "description": "A string field"},
|
||||
{"name": "field2", "type": "int", "default": 0, "description": "An integer field"},
|
||||
{"name": "field3", "type": "bool", "default": False, "description": "A boolean field"},
|
||||
]
|
||||
model = build_model_from_schema(schema)
|
||||
instance = model(field1="test", field2=123, field3=True)
|
||||
assert instance.field1 == "test"
|
||||
assert instance.field2 == 123
|
||||
assert instance.field3 is True
|
||||
|
||||
# Handles empty schema gracefully without errors
|
||||
def test_handle_empty_schema(self):
|
||||
schema = []
|
||||
model = build_model_from_schema(schema)
|
||||
instance = model()
|
||||
assert instance is not None
|
||||
|
||||
# Ensure the model created from schema has the expected attributes by checking on an instance
|
||||
def test_handles_multiple_fields_fixed_with_instance_check(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1"},
|
||||
{"name": "field2", "type": "int", "default": 42},
|
||||
{"name": "field3", "type": "list", "default": [1, 2, 3]},
|
||||
{"name": "field4", "type": "dict", "default": {"key": "value"}},
|
||||
]
|
||||
|
||||
model = build_model_from_schema(schema)
|
||||
model_instance = model(field1="test", field2=123, field3=[1, 2, 3], field4={"key": "value"})
|
||||
|
||||
assert issubclass(model, BaseModel)
|
||||
assert hasattr(model_instance, "field1")
|
||||
assert hasattr(model_instance, "field2")
|
||||
assert hasattr(model_instance, "field3")
|
||||
assert hasattr(model_instance, "field4")
|
||||
|
||||
# Correctly accesses descriptions using the recommended fix
|
||||
def test_correctly_accesses_descriptions_recommended_fix(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1", "description": "Description for field1"},
|
||||
{"name": "field2", "type": "int", "default": 42, "description": "Description for field2"},
|
||||
{"name": "field3", "type": "list", "default": [1, 2, 3], "description": "Description for field3"},
|
||||
{"name": "field4", "type": "dict", "default": {"key": "value"}, "description": "Description for field4"},
|
||||
]
|
||||
|
||||
model = build_model_from_schema(schema)
|
||||
|
||||
assert model.model_fields["field1"].description == "Description for field1"
|
||||
assert model.model_fields["field2"].description == "Description for field2"
|
||||
assert model.model_fields["field3"].description == "Description for field3"
|
||||
assert model.model_fields["field4"].description == "Description for field4"
|
||||
|
||||
# Supports both single and multiple type annotations
|
||||
def test_supports_single_and_multiple_type_annotations(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1", "description": "Description 1"},
|
||||
{"name": "field2", "type": "list", "default": [1, 2, 3], "description": "Description 2", "multiple": True},
|
||||
{"name": "field3", "type": "int", "default": 100, "description": "Description 3"},
|
||||
]
|
||||
model_type = build_model_from_schema(schema)
|
||||
assert issubclass(model_type, BaseModel)
|
||||
|
||||
# Manages unknown field types by defaulting to Any
|
||||
def test_manages_unknown_field_types(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1"},
|
||||
{"name": "field2", "type": "unknown_type", "default": "default_value2"},
|
||||
]
|
||||
with pytest.raises(ValueError):
|
||||
build_model_from_schema(schema)
|
||||
|
||||
# Confirms that the function raises a specific exception for invalid input
|
||||
def test_raises_error_for_invalid_input_different_exception_with_specific_exception(self):
|
||||
with pytest.raises(ValueError):
|
||||
schema = [{"name": "field1", "type": "invalid_type", "default": "default_value"}]
|
||||
build_model_from_schema(schema)
|
||||
|
||||
# Processes schemas with missing optional keys like description or multiple
|
||||
def test_process_schema_missing_optional_keys_updated(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1"},
|
||||
{"name": "field2", "type": "int", "default": 0, "description": "Field 2 description"},
|
||||
{"name": "field3", "type": "list", "default": [], "multiple": True},
|
||||
{"name": "field4", "type": "dict", "default": {}, "description": "Field 4 description", "multiple": True},
|
||||
]
|
||||
result_model = build_model_from_schema(schema)
|
||||
assert result_model.__annotations__["field1"] == str # noqa: E721
|
||||
assert result_model.model_fields["field1"].description == ""
|
||||
assert result_model.__annotations__["field2"] == int # noqa: E721
|
||||
assert result_model.model_fields["field2"].description == "Field 2 description"
|
||||
assert result_model.__annotations__["field3"] == list[list[Any]]
|
||||
assert result_model.model_fields["field3"].description == ""
|
||||
assert result_model.__annotations__["field4"] == list[dict[str, Any]]
|
||||
assert result_model.model_fields["field4"].description == "Field 4 description"
|
||||
|
||||
# Deals with schemas containing fields with None as default values
|
||||
def test_schema_fields_with_none_default(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": None, "description": "Field 1 description"},
|
||||
{"name": "field2", "type": "int", "default": None, "description": "Field 2 description"},
|
||||
{"name": "field3", "type": "list", "default": None, "description": "Field 3 description", "multiple": True},
|
||||
]
|
||||
model = build_model_from_schema(schema)
|
||||
assert model.model_fields["field1"].default == PydanticUndefined # noqa: E711
|
||||
assert model.model_fields["field2"].default == PydanticUndefined # noqa: E711
|
||||
assert model.model_fields["field3"].default == PydanticUndefined # noqa: E711
|
||||
|
||||
# Checks for proper handling of nested list and dict types
|
||||
def test_nested_list_and_dict_types_handling(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "list", "default": [], "description": "list field", "multiple": True},
|
||||
{"name": "field2", "type": "dict", "default": {}, "description": "Dict field"},
|
||||
]
|
||||
model_type = build_model_from_schema(schema)
|
||||
assert issubclass(model_type, BaseModel)
|
||||
|
||||
# Verifies that the function can handle large schemas efficiently
|
||||
def test_handle_large_schemas_efficiently(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1", "description": "Description 1"},
|
||||
{"name": "field2", "type": "int", "default": 100, "description": "Description 2"},
|
||||
{"name": "field3", "type": "list", "default": [1, 2, 3], "description": "Description 3", "multiple": True},
|
||||
{"name": "field4", "type": "dict", "default": {"key": "value"}, "description": "Description 4"},
|
||||
]
|
||||
model_type = build_model_from_schema(schema)
|
||||
assert issubclass(model_type, BaseModel)
|
||||
|
||||
# Ensures that the function returns a valid Pydantic model class
|
||||
def test_returns_valid_model_class(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1", "description": "Description for field1"},
|
||||
{"name": "field2", "type": "int", "default": 42, "description": "Description for field2", "multiple": True},
|
||||
]
|
||||
model_class = build_model_from_schema(schema)
|
||||
assert issubclass(model_class, BaseModel)
|
||||
|
||||
# Validates that the last occurrence of a duplicate field name defines the type in the schema
|
||||
def test_no_duplicate_field_names_fixed_fixed(self):
|
||||
schema = [
|
||||
{"name": "field1", "type": "str", "default": "default_value1"},
|
||||
{"name": "field2", "type": "int", "default": 0},
|
||||
{"name": "field1", "type": "float", "default": 0.0}, # Duplicate field name
|
||||
]
|
||||
model = build_model_from_schema(schema)
|
||||
assert model.__annotations__["field1"] == float # noqa: E721
|
||||
assert model.__annotations__["field2"] == int # noqa: E721
|
||||
Loading…
Add table
Add a link
Reference in a new issue