Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 12 additions & 6 deletions langfuse/_utils/serializer.py
Original file line number Diff line number Diff line change
Expand Up @@ -70,13 +70,17 @@ def _default_inner(self, obj: Any) -> Any:

# Check if numpy is available and if the object is a numpy scalar
# If so, convert it to a Python scalar using the item() method
# and recurse through default() so non-finite floats and unsafe
# integers are normalized
if np is not None and isinstance(obj, np.generic):
return obj.item()
return self.default(obj.item())

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 NumPy Scalars Recurse Indefinitely

Extended-precision scalars such as np.longdouble and np.clongdouble may return themselves from .item(). Passing that result back to self.default() repeatedly re-enters this branch without reaching the later depth guard, eventually hitting Python's recursion limit instead of producing a useful serialized value.

Prompt To Fix With AI
This is a comment left during a code review.
Path: langfuse/_utils/serializer.py
Line: 76

Comment:
**NumPy Scalars Recurse Indefinitely**

Extended-precision scalars such as `np.longdouble` and `np.clongdouble` may return themselves from `.item()`. Passing that result back to `self.default()` repeatedly re-enters this branch without reaching the later depth guard, eventually hitting Python's recursion limit instead of producing a useful serialized value.

---

For each issue above, determine whether it is valid and should be fixed. If so, fix it directly.


# Check if numpy is available and if the object is a numpy array
# If so, convert it to a Python list using the tolist() method
# and recurse through default() so nested non-finite floats and
# unsafe integers are normalized
if np is not None and isinstance(obj, np.ndarray):
return obj.tolist()
return self.default(obj.tolist())

if isinstance(obj, float) and math.isnan(obj):
return "NaN"
Expand All @@ -93,7 +97,7 @@ def _default_inner(self, obj: Any) -> Any:
return str(obj)

if isinstance(obj, enum.Enum):
return obj.value
return self.default(obj.value)

if isinstance(obj, Queue):
return type(obj).__name__
Expand Down Expand Up @@ -127,7 +131,9 @@ def _default_inner(self, obj: Any) -> Any:
return f"<{type(obj).__name__}>"

if is_dataclass(obj):
return asdict(obj) # type: ignore
# Recurse the dict produced by asdict() back through default()
# so nested non-finite floats and unsafe integers are normalized

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Dataclass Keys Defeat Skipkeys

A dataclass field containing a mapping with tuple keys is now recursively processed before the standard encoder can apply skipkeys=True. Each tuple key becomes an unhashable list while rebuilding the dictionary, so the entire mapping is replaced by the serializer's error marker instead of omitting the unsupported key.

Prompt To Fix With AI
This is a comment left during a code review.
Path: langfuse/_utils/serializer.py
Line: 135

Comment:
**Dataclass Keys Defeat Skipkeys**

A dataclass field containing a mapping with tuple keys is now recursively processed before the standard encoder can apply `skipkeys=True`. Each tuple key becomes an unhashable list while rebuilding the dictionary, so the entire mapping is replaced by the serializer's error marker instead of omitting the unsupported key.

---

For each issue above, determine whether it is valid and should be fixed. If so, fix it directly.

return self.default(asdict(obj)) # type: ignore

if isinstance(obj, BaseModel):
obj.model_rebuild()
Expand All @@ -144,10 +150,10 @@ def _default_inner(self, obj: Any) -> Any:

# if langchain is not available, the Serializable type is NoneType
if Serializable is not type(None) and isinstance(obj, Serializable): # type: ignore
return obj.to_json()
return self.default(obj.to_json())

if isinstance(obj, (tuple, set, frozenset)):
return list(obj)
return [self.default(item) for item in obj]

if isinstance(obj, dict):
return {self.default(k): self.default(v) for k, v in obj.items()}
Expand Down
93 changes: 93 additions & 0 deletions tests/unit/test_serializer.py
Original file line number Diff line number Diff line change
Expand Up @@ -362,3 +362,96 @@ def test_dict_with_non_string_keys_is_serialized(input_obj, expected):
result = json.loads(EventSerializer().encode(input_obj))

assert result == expected


def _reject_json_constant(token: str) -> None:
# json.loads accepts bare NaN/Infinity by default, so reject them the way
# the ingestion server's strict JSON parser does.
raise ValueError(f"invalid JSON constant emitted: {token}")


def test_non_finite_floats_in_tuple_set_frozenset():
# Values inside tuple/set/frozenset must be recursed through default() like
# list items, so non-finite floats are converted to safe string tokens
# instead of escaping as bare NaN/Infinity (invalid JSON).
serializer = EventSerializer()

assert json.loads(
serializer.encode((float("nan"),)),
parse_constant=_reject_json_constant,
) == ["NaN"]
assert json.loads(
serializer.encode({float("inf")}), parse_constant=_reject_json_constant
) == ["Infinity"]
assert json.loads(
serializer.encode(frozenset([float("-inf")])),
parse_constant=_reject_json_constant,
) == ["-Infinity"]
assert json.loads(
serializer.encode({"t": (1, float("nan"))}),
parse_constant=_reject_json_constant,
) == {"t": [1, "NaN"]}


def test_js_unsafe_integers_in_tuple_set_frozenset():
# Integers outside the JavaScript safe range must be stringified even when
# nested in tuple/set/frozenset containers.
serializer = EventSerializer()
unsafe = 2**60

assert json.loads(serializer.encode((unsafe,))) == [str(unsafe)]
assert json.loads(serializer.encode({unsafe})) == [str(unsafe)]
assert json.loads(serializer.encode(frozenset([unsafe]))) == [str(unsafe)]


def test_dataclass_with_non_finite_float():
# asdict() output must be recursed through default() so nested non-finite
# floats are converted to safe string tokens.
@dataclass
class WithFloats:
finite: float
nan: float

serializer = EventSerializer()
parsed = json.loads(
serializer.encode(WithFloats(finite=1.5, nan=float("nan"))),
parse_constant=_reject_json_constant,
)
assert parsed == {"finite": 1.5, "nan": "NaN"}


def test_enum_with_non_finite_float_value():
class FloatEnum(Enum):
NAN = float("nan")

serializer = EventSerializer()
parsed = json.loads(
serializer.encode(FloatEnum.NAN), parse_constant=_reject_json_constant
)
assert parsed == "NaN"


def test_numpy_types_with_non_finite_and_unsafe_values():
np = pytest.importorskip("numpy")

serializer = EventSerializer()

# np.generic scalar: .item() result must be recursed through default()
assert (
json.loads(
serializer.encode(np.float64("nan")),
parse_constant=_reject_json_constant,
)
== "NaN"
)
assert json.loads(serializer.encode(np.int64(2**60))) == str(2**60)

# np.ndarray: tolist() result must be recursed through default()
parsed = json.loads(
serializer.encode(np.array([float("nan"), float("inf"), 1.0])),
parse_constant=_reject_json_constant,
)
assert parsed == ["NaN", "Infinity", 1.0]
assert json.loads(serializer.encode(np.array([2**60], dtype=np.int64))) == [
str(2**60)
]