From 2bbff986fdf1bdd56dbc1402392a297b21f22f04 Mon Sep 17 00:00:00 2001 From: jahngalt <26158054+jahngalt@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:09:43 +0300 Subject: [PATCH] fix(langfuse_otel): fall back to escaped JSON on unpaired surrogates; cover output branches safe_dumps(ensure_ascii=False) now checks that its result encodes as UTF-8 and otherwise returns the escaped ensure_ascii=True output, so a lone surrogate in the text cannot break OTLP export. The structure is built once. Adds raw-attribute tests for the tool_calls and output-items branches and surrogate tests for safe_dumps and the Langfuse OTEL attributes. --- litellm/litellm_core_utils/safe_json_dumps.py | 18 ++- tests/unit/integrations/test_langfuse_otel.py | 113 ++++++++++++++++++ .../test_safe_json_dumps.py | 23 ++++ 3 files changed, 148 insertions(+), 6 deletions(-) diff --git a/litellm/litellm_core_utils/safe_json_dumps.py b/litellm/litellm_core_utils/safe_json_dumps.py index 5cd892b26cc..a8227975c0f 100644 --- a/litellm/litellm_core_utils/safe_json_dumps.py +++ b/litellm/litellm_core_utils/safe_json_dumps.py @@ -93,10 +93,16 @@ def safe_dumps( """Serialize data to JSON text through safe_json_structure. ensure_ascii=False keeps non-ASCII characters as-is instead of \\uXXXX - escapes; the parsed value is identical either way. + escapes; the parsed value is identical either way. The result of an + ensure_ascii=False call is always encodable as UTF-8: if the text holds + something that is not (e.g. an unpaired surrogate such as "\\ud800"), the + escaped ensure_ascii=True output is returned instead. """ - return json.dumps( - safe_json_structure(data, max_depth, value_transform), - default=str, - ensure_ascii=ensure_ascii, - ) + structure: Final = safe_json_structure(data, max_depth, value_transform) + text: Final = json.dumps(structure, default=str, ensure_ascii=ensure_ascii) + if not ensure_ascii: + try: + text.encode("utf-8") + except UnicodeEncodeError: + return json.dumps(structure, default=str, ensure_ascii=True) + return text diff --git a/tests/unit/integrations/test_langfuse_otel.py b/tests/unit/integrations/test_langfuse_otel.py index 1ce7a85c94e..3b91853e692 100644 --- a/tests/unit/integrations/test_langfuse_otel.py +++ b/tests/unit/integrations/test_langfuse_otel.py @@ -398,6 +398,119 @@ class TestLangfuseOtelIntegration: assert "\\u" not in input_raw and "\\u" not in output_raw assert json.loads(input_raw) == kwargs["messages"] + @staticmethod + def _raw_attributes(kwargs, response_obj): + """Run the attribute setter and return the raw {attribute: value} it wrote.""" + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, response_obj + ) + return { + call.args[1]: call.args[2] + for call in mock_safe_set_attribute.call_args_list + } + + def test_set_langfuse_specific_attributes_tool_calls_keep_non_ascii_unescaped(self): + """Tool call arguments with non-ASCII text are written as-is, not as \\uXXXX escapes.""" + from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + from litellm.types.utils import ( + ChatCompletionMessageToolCall, + Choices, + Function, + ModelResponse, + ) + + response_obj = ModelResponse( + id="chatcmpl-test", + model="gpt-4o", + choices=[ + Choices( + finish_reason="tool_calls", + message={ + "role": "assistant", + "content": None, + "tool_calls": [ + ChatCompletionMessageToolCall( + function=Function( + arguments='{"greeting": "Привет, мир", "city": "東京"}', + name="say_hello", + ), + id="call_123", + type="function", + ) + ], + }, + ) + ], + ) + + raw = self._raw_attributes({}, response_obj) + + output_raw = raw[LangfuseSpanAttributes.OBSERVATION_OUTPUT.value] + assert "Привет, мир" in output_raw and "東京" in output_raw + assert "\\u" not in output_raw + assert json.loads(output_raw) == [ + { + "id": "chatcmpl-test", + "name": "say_hello", + "arguments": {"greeting": "Привет, мир", "city": "東京"}, + "call_id": "call_123", + "type": "function_call", + } + ] + + def test_set_langfuse_specific_attributes_output_items_keep_non_ascii_unescaped( + self, + ): + """Responses API output items with non-ASCII text are written as-is, not as \\uXXXX escapes.""" + from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + response_obj = ResponsesAPIResponse( + id="response-unicode", + created_at=1625247600, + output=[ + ResponseOutputMessage( + id="msg-001", + type="message", + role="assistant", + status="completed", + content=[ + ResponseOutputText( + annotations=[], + text="こんにちは, Ünïcödé", + type="output_text", + ) + ], + ) + ], + ) + + raw = self._raw_attributes({"call_type": "responses"}, response_obj) + + output_raw = raw[LangfuseSpanAttributes.OBSERVATION_OUTPUT.value] + assert "こんにちは, Ünïcödé" in output_raw + assert "\\u" not in output_raw + assert json.loads(output_raw) == [ + {"role": "assistant", "content": "こんにちは, Ünïcödé"} + ] + + def test_set_langfuse_specific_attributes_unpaired_surrogate_stays_utf8_encodable( + self, + ): + """An unpaired surrogate must not yield a string that cannot be UTF-8 encoded (OTLP export).""" + from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + + kwargs = {"messages": [{"role": "user", "content": "Привет \ud800 мир"}]} + + raw = self._raw_attributes(kwargs, None) + + input_raw = raw[LangfuseSpanAttributes.OBSERVATION_INPUT.value] + input_raw.encode("utf-8") + assert json.loads(input_raw) == kwargs["messages"] + def test_set_langfuse_specific_attributes_with_tool_calls(self): """Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format.""" from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes diff --git a/tests/unit/litellm_core_utils/test_safe_json_dumps.py b/tests/unit/litellm_core_utils/test_safe_json_dumps.py index 8f9233f2626..e20cf15be54 100644 --- a/tests/unit/litellm_core_utils/test_safe_json_dumps.py +++ b/tests/unit/litellm_core_utils/test_safe_json_dumps.py @@ -248,3 +248,26 @@ def test_ensure_ascii_false_keeps_non_ascii(): result = safe_dumps(data, ensure_ascii=False) assert result == '{"text": "Привет, 世界", "nested": ["ü", {"k": "é"}]}' assert json.loads(result) == json.loads(safe_dumps(data)) + + +def test_ensure_ascii_false_unpaired_surrogate_falls_back_to_escaped_json(): + data = {"t": "a\ud800b"} + result = safe_dumps(data, ensure_ascii=False) + result.encode("utf-8") + assert result == '{"t": "a\\ud800b"}' + assert json.loads(result) == data + + +def test_ensure_ascii_false_surrogate_fallback_escapes_all_non_ascii(): + # The fallback is the previous escaped output as a whole, not a partial repair + data = {"t": "Привет \ud800", "u": "こんにちは"} + result = safe_dumps(data, ensure_ascii=False) + result.encode("utf-8") + assert result == safe_dumps(data) + assert json.loads(result) == data + + +def test_ensure_ascii_false_valid_non_ascii_not_escaped_alongside_ascii_only_default(): + result = safe_dumps({"t": "Ünïcödé"}, ensure_ascii=False) + assert result == '{"t": "Ünïcödé"}' + result.encode("utf-8")