fix(langfuse_otel): fall back to escaped JSON on unpaired surrogates; cover output branches

safe_dumps(ensure_ascii=False) now checks that its result encodes as UTF-8
and otherwise returns the escaped ensure_ascii=True output, so a lone
surrogate in the text cannot break OTLP export. The structure is built once.
Adds raw-attribute tests for the tool_calls and output-items branches and
surrogate tests for safe_dumps and the Langfuse OTEL attributes.
This commit is contained in:
jahngalt 2026-09-29 10:09:43 +03:00
parent 13eb5bf702
commit 2bbff986fd
3 changed files with 148 additions and 6 deletions

View file

@ -93,10 +93,16 @@ def safe_dumps(
"""Serialize data to JSON text through safe_json_structure.
ensure_ascii=False keeps non-ASCII characters as-is instead of \\uXXXX
escapes; the parsed value is identical either way.
escapes; the parsed value is identical either way. The result of an
ensure_ascii=False call is always encodable as UTF-8: if the text holds
something that is not (e.g. an unpaired surrogate such as "\\ud800"), the
escaped ensure_ascii=True output is returned instead.
"""
return json.dumps(
safe_json_structure(data, max_depth, value_transform),
default=str,
ensure_ascii=ensure_ascii,
)
structure: Final = safe_json_structure(data, max_depth, value_transform)
text: Final = json.dumps(structure, default=str, ensure_ascii=ensure_ascii)
if not ensure_ascii:
try:
text.encode("utf-8")
except UnicodeEncodeError:
return json.dumps(structure, default=str, ensure_ascii=True)
return text

View file

@ -398,6 +398,119 @@ class TestLangfuseOtelIntegration:
assert "\\u" not in input_raw and "\\u" not in output_raw
assert json.loads(input_raw) == kwargs["messages"]
@staticmethod
def _raw_attributes(kwargs, response_obj):
"""Run the attribute setter and return the raw {attribute: value} it wrote."""
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), kwargs, response_obj
)
return {
call.args[1]: call.args[2]
for call in mock_safe_set_attribute.call_args_list
}
def test_set_langfuse_specific_attributes_tool_calls_keep_non_ascii_unescaped(self):
"""Tool call arguments with non-ASCII text are written as-is, not as \\uXXXX escapes."""
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
from litellm.types.utils import (
ChatCompletionMessageToolCall,
Choices,
Function,
ModelResponse,
)
response_obj = ModelResponse(
id="chatcmpl-test",
model="gpt-4o",
choices=[
Choices(
finish_reason="tool_calls",
message={
"role": "assistant",
"content": None,
"tool_calls": [
ChatCompletionMessageToolCall(
function=Function(
arguments='{"greeting": "Привет, мир", "city": "東京"}',
name="say_hello",
),
id="call_123",
type="function",
)
],
},
)
],
)
raw = self._raw_attributes({}, response_obj)
output_raw = raw[LangfuseSpanAttributes.OBSERVATION_OUTPUT.value]
assert "Привет, мир" in output_raw and "東京" in output_raw
assert "\\u" not in output_raw
assert json.loads(output_raw) == [
{
"id": "chatcmpl-test",
"name": "say_hello",
"arguments": {"greeting": "Привет, мир", "city": "東京"},
"call_id": "call_123",
"type": "function_call",
}
]
def test_set_langfuse_specific_attributes_output_items_keep_non_ascii_unescaped(
self,
):
"""Responses API output items with non-ASCII text are written as-is, not as \\uXXXX escapes."""
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
response_obj = ResponsesAPIResponse(
id="response-unicode",
created_at=1625247600,
output=[
ResponseOutputMessage(
id="msg-001",
type="message",
role="assistant",
status="completed",
content=[
ResponseOutputText(
annotations=[],
text="こんにちは, Ünïcödé",
type="output_text",
)
],
)
],
)
raw = self._raw_attributes({"call_type": "responses"}, response_obj)
output_raw = raw[LangfuseSpanAttributes.OBSERVATION_OUTPUT.value]
assert "こんにちは, Ünïcödé" in output_raw
assert "\\u" not in output_raw
assert json.loads(output_raw) == [
{"role": "assistant", "content": "こんにちは, Ünïcödé"}
]
def test_set_langfuse_specific_attributes_unpaired_surrogate_stays_utf8_encodable(
self,
):
"""An unpaired surrogate must not yield a string that cannot be UTF-8 encoded (OTLP export)."""
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
kwargs = {"messages": [{"role": "user", "content": "Привет \ud800 мир"}]}
raw = self._raw_attributes(kwargs, None)
input_raw = raw[LangfuseSpanAttributes.OBSERVATION_INPUT.value]
input_raw.encode("utf-8")
assert json.loads(input_raw) == kwargs["messages"]
def test_set_langfuse_specific_attributes_with_tool_calls(self):
"""Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format."""
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes

View file

@ -248,3 +248,26 @@ def test_ensure_ascii_false_keeps_non_ascii():
result = safe_dumps(data, ensure_ascii=False)
assert result == '{"text": "Привет, 世界", "nested": ["ü", {"k": "é"}]}'
assert json.loads(result) == json.loads(safe_dumps(data))
def test_ensure_ascii_false_unpaired_surrogate_falls_back_to_escaped_json():
data = {"t": "a\ud800b"}
result = safe_dumps(data, ensure_ascii=False)
result.encode("utf-8")
assert result == '{"t": "a\\ud800b"}'
assert json.loads(result) == data
def test_ensure_ascii_false_surrogate_fallback_escapes_all_non_ascii():
# The fallback is the previous escaped output as a whole, not a partial repair
data = {"t": "Привет \ud800", "u": "こんにちは"}
result = safe_dumps(data, ensure_ascii=False)
result.encode("utf-8")
assert result == safe_dumps(data)
assert json.loads(result) == data
def test_ensure_ascii_false_valid_non_ascii_not_escaped_alongside_ascii_only_default():
result = safe_dumps({"t": "Ünïcödé"}, ensure_ascii=False)
assert result == '{"t": "Ünïcödé"}'
result.encode("utf-8")