refactor: Implement unified serialization function (#6044)

* feat: Implement serialization functions for various data types and add a unified serialize method

* feat: Enhance serialization by adding support for primitive types, enums, and generic types

* fix: Update Pinecone integration to use VectorStore and handle import errors gracefully

* test: Add hypothesis-based tests for serialization functions across various data types

* refactor: Replace custom serialization logic with unified serialize function for consistency and maintainability

* refactor: Replace recursive serialization function with unified serialize method for improved clarity and maintainability

* refactor: Replace custom serialization logic with unified serialize function for improved consistency and clarity

* refactor: Enhance serialization logic by adding instance handling and streamlining type checks

* refactor: Remove custom dictionary serialization from ResultDataResponse for streamlined handling

* refactor: Enhance serialization in ResultDataResponse by adding max_items_length for improved handling of outputs, logs, messages, and artifacts

* refactor: Move MAX_ITEMS_LENGTH and MAX_TEXT_LENGTH constants to serialization module for better organization

* refactor: Simplify message serialization in Log model by utilizing unified serialize function

* refactor: Remove unnecessary pytest marker from TestSerializationHypothesis class

* optimize _serialize_bytes

Co-authored-by: codeflash-ai[bot] <148906541+codeflash-ai[bot]@users.noreply.github.com>

* feat: Add support for numpy integer type serialization

* feat: Enhance serialization with support for pandas and numpy types

* test: Add comprehensive serialization tests for numpy and pandas types

* fix: Update _serialize_dispatcher to return string representation for unsupported types

* fix: Update _serialize_dispatcher to return the object directly instead of its string representation

* optmize conditional

Co-authored-by: codeflash-ai[bot] <148906541+codeflash-ai[bot]@users.noreply.github.com>

* optimize length check

Co-authored-by: codeflash-ai[bot] <148906541+codeflash-ai[bot]@users.noreply.github.com>

* fix: Update string and list truncation to include ellipsis for clarity

* fix: Update _serialize_primitive to exclude string type from primitive handling

* feat: Enhance serialization to handle numpy types and introduce unserializable sentinel

* fix: Update test cases for serialization of numpy boolean values for consistency

---------

Co-authored-by: codeflash-ai[bot] <148906541+codeflash-ai[bot]@users.noreply.github.com>
This commit is contained in:
Gabriel Luiz Freitas Almeida 2025-02-03 12:12:03 -03:00 • committed by GitHub
commit c73070cd52
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
20 changed files with 696 additions and 186 deletions

View file

@ -4,6 +4,7 @@ from hypothesis import HealthCheck, example, given, settings
from hypothesis import strategies as st
from langflow.api.v1.schemas import ResultDataResponse, VertexBuildResponse
from langflow.schema.schema import OutputValue
from langflow.serialization import serialize
from langflow.services.tracing.schema import Log
from pydantic import BaseModel
@ -26,9 +27,9 @@ def test_result_data_response_truncation(long_string):
)
response.serialize_model()
truncated = response._serialize_and_truncate(long_string, max_length=TEST_TEXT_LENGTH)
assert len(truncated) <= TEST_TEXT_LENGTH + len("... [truncated]")
assert "... [truncated]" in truncated
truncated = serialize(long_string, max_length=TEST_TEXT_LENGTH)
assert len(truncated) <= TEST_TEXT_LENGTH + len("...")
assert "..." in truncated
@given(
@ -77,20 +78,20 @@ def test_result_data_response_nested_structures(long_list, long_dict):
"dict": long_dict,
}
response = ResultDataResponse(results=nested_data)
serialized = response._serialize_and_truncate(nested_data, max_length=TEST_TEXT_LENGTH)
ResultDataResponse(results=nested_data)
serialized = serialize(nested_data, max_length=TEST_TEXT_LENGTH)
# Check list items
for item in serialized["list"]:
assert len(item) <= TEST_TEXT_LENGTH + len("... [truncated]")
assert len(item) <= TEST_TEXT_LENGTH + len("...")
if len(item) > TEST_TEXT_LENGTH:
assert "... [truncated]" in item
assert "..." in item
# Check dict values
for val in serialized["dict"].values():
assert len(val) <= TEST_TEXT_LENGTH + len("... [truncated]")
assert len(val) <= TEST_TEXT_LENGTH + len("...")
if len(val) > TEST_TEXT_LENGTH:
assert "... [truncated]" in val
assert "..." in val
@given(
@ -114,7 +115,7 @@ def test_result_data_response_outputs(outputs_dict):
outputs = {key: OutputValue(type="text", message=value) for key, value in outputs_dict.items()}
response = ResultDataResponse(outputs=outputs)
serialized = ResultDataResponse._serialize_and_truncate(response, max_length=TEST_TEXT_LENGTH)
serialized = serialize(response, max_length=TEST_TEXT_LENGTH)
# Check outputs are properly serialized and truncated
for key, value in outputs_dict.items():
@ -124,9 +125,9 @@ def test_result_data_response_outputs(outputs_dict):
# Check message truncation
message = serialized_output["message"]
assert len(message) <= TEST_TEXT_LENGTH + len("... [truncated]"), f"Message length: {len(message)}"
assert len(message) <= TEST_TEXT_LENGTH + len("..."), f"Message length: {len(message)}"
if len(value) > TEST_TEXT_LENGTH:
assert "... [truncated]" in message
assert "..." in message
assert message.startswith(value[:TEST_TEXT_LENGTH])
else:
assert message == value
@ -158,7 +159,7 @@ def test_result_data_response_logs(log_messages):
}
response = ResultDataResponse(logs=logs)
serialized = ResultDataResponse._serialize_and_truncate(response, max_length=TEST_TEXT_LENGTH)
serialized = serialize(response, max_length=TEST_TEXT_LENGTH)
# Check logs are properly serialized and truncated
assert "test_node" in serialized["logs"]
@ -171,9 +172,9 @@ def test_result_data_response_logs(log_messages):
# Check message truncation
message = serialized_log["message"]
assert len(message) <= TEST_TEXT_LENGTH + len("... [truncated]")
assert len(message) <= TEST_TEXT_LENGTH + len("...")
if len(log_msg) > TEST_TEXT_LENGTH:
assert "... [truncated]" in message
assert "..." in message
assert message.startswith(log_msg[:TEST_TEXT_LENGTH])
else:
assert message == log_msg
@ -225,7 +226,7 @@ def test_result_data_response_combined_fields(outputs_dict, log_messages):
message={"text": "test"},
artifacts={"file": "test.txt"},
)
serialized = ResultDataResponse._serialize_and_truncate(response, max_length=TEST_TEXT_LENGTH)
serialized = serialize(response, max_length=TEST_TEXT_LENGTH)
# Check all fields are present
assert "outputs" in serialized
@ -243,8 +244,8 @@ def test_result_data_response_combined_fields(outputs_dict, log_messages):
# Check message truncation
message = serialized_output["message"]
if len(value) > TEST_TEXT_LENGTH:
assert len(message) <= TEST_TEXT_LENGTH + len("... [truncated]")
assert "... [truncated]" in message
assert len(message) <= TEST_TEXT_LENGTH + len("...")
assert "..." in message
else:
assert message == value
@ -260,8 +261,8 @@ def test_result_data_response_combined_fields(outputs_dict, log_messages):
# Check message truncation
message = serialized_log["message"]
if len(log_msg) > TEST_TEXT_LENGTH:
assert len(message) <= TEST_TEXT_LENGTH + len("... [truncated]")
assert "... [truncated]" in message
assert len(message) <= TEST_TEXT_LENGTH + len("...")
assert "..." in message
else:
assert message == log_msg
@ -311,6 +312,6 @@ def test_vertex_build_response_with_long_data(long_string):
)
response.model_dump()
truncated = result_data._serialize_and_truncate(long_string, max_length=TEST_TEXT_LENGTH)
assert len(truncated) <= TEST_TEXT_LENGTH + len("... [truncated]")
assert "... [truncated]" in truncated
truncated = serialize(long_string, max_length=TEST_TEXT_LENGTH)
assert len(truncated) <= TEST_TEXT_LENGTH + len("...")
assert "..." in truncated