From 13eb5bf702f87820a7d23bed1a6f765cfdbcda71 Mon Sep 17 00:00:00 2001 From: jahngalt <26158054+jahngalt@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:55:10 +0300 Subject: [PATCH 1/2] fix(langfuse_otel): keep non-ASCII input/output unescaped safe_dumps gains an ensure_ascii flag (default True, unchanged behavior). The Langfuse OTEL integration passes ensure_ascii=False so observation input/output keep non-ASCII text as-is instead of \uXXXX escapes. --- .../integrations/langfuse/langfuse_otel.py | 8 ++--- litellm/litellm_core_utils/safe_json_dumps.py | 13 +++++-- tests/unit/integrations/test_langfuse_otel.py | 36 +++++++++++++++++++ .../test_safe_json_dumps.py | 12 +++++++ 4 files changed, 63 insertions(+), 6 deletions(-) diff --git a/litellm/integrations/langfuse/langfuse_otel.py b/litellm/integrations/langfuse/langfuse_otel.py index a96fac32c2a..4f74c49eac4 100644 --- a/litellm/integrations/langfuse/langfuse_otel.py +++ b/litellm/integrations/langfuse/langfuse_otel.py @@ -159,7 +159,7 @@ class LangfuseOtelLogger(OpenTelemetry): safe_set_attribute( span, LangfuseSpanAttributes.OBSERVATION_OUTPUT.value, - safe_dumps(transformed_tool_calls), + safe_dumps(transformed_tool_calls, ensure_ascii=False), ) else: output_data: Final = {} @@ -171,7 +171,7 @@ class LangfuseOtelLogger(OpenTelemetry): safe_set_attribute( span, LangfuseSpanAttributes.OBSERVATION_OUTPUT.value, - safe_dumps(output_data), + safe_dumps(output_data, ensure_ascii=False), ) output: Final = response_obj.get("output", []) @@ -215,7 +215,7 @@ class LangfuseOtelLogger(OpenTelemetry): safe_set_attribute( span, LangfuseSpanAttributes.OBSERVATION_OUTPUT.value, - safe_dumps(output_items_data), + safe_dumps(output_items_data, ensure_ascii=False), ) @staticmethod @@ -250,7 +250,7 @@ class LangfuseOtelLogger(OpenTelemetry): safe_set_attribute( span, LangfuseSpanAttributes.OBSERVATION_INPUT.value, - safe_dumps(messages), + safe_dumps(messages, ensure_ascii=False), ) LangfuseOtelLogger._set_observation_output(span=span, response_obj=response_obj) diff --git a/litellm/litellm_core_utils/safe_json_dumps.py b/litellm/litellm_core_utils/safe_json_dumps.py index 63242a580e7..5cd892b26cc 100644 --- a/litellm/litellm_core_utils/safe_json_dumps.py +++ b/litellm/litellm_core_utils/safe_json_dumps.py @@ -88,6 +88,15 @@ def safe_dumps( data: object, max_depth: int = DEFAULT_MAX_RECURSE_DEPTH, value_transform: Callable[[str | None, str], str] | None = None, + ensure_ascii: bool = True, ) -> str: - """Serialize data to JSON text through safe_json_structure.""" - return json.dumps(safe_json_structure(data, max_depth, value_transform), default=str) + """Serialize data to JSON text through safe_json_structure. + + ensure_ascii=False keeps non-ASCII characters as-is instead of \\uXXXX + escapes; the parsed value is identical either way. + """ + return json.dumps( + safe_json_structure(data, max_depth, value_transform), + default=str, + ensure_ascii=ensure_ascii, + ) diff --git a/tests/unit/integrations/test_langfuse_otel.py b/tests/unit/integrations/test_langfuse_otel.py index 0a9ce55fe16..1ce7a85c94e 100644 --- a/tests/unit/integrations/test_langfuse_otel.py +++ b/tests/unit/integrations/test_langfuse_otel.py @@ -362,6 +362,42 @@ class TestLangfuseOtelIntegration: actual == expect_output ), "Mismatch in observation input/output OTEL attributes." + def test_set_langfuse_specific_attributes_keeps_non_ascii_unescaped(self): + """Non-ASCII input/output is written as-is, not as \\uXXXX escapes.""" + from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + from litellm.types.utils import Choices, ModelResponse + + response_obj = ModelResponse( + id="chatcmpl-test", + model="gpt-4o", + choices=[ + Choices( + finish_reason="stop", + message={"role": "assistant", "content": "В Токио солнечно. 晴れ"}, + ) + ], + ) + kwargs = {"messages": [{"role": "user", "content": "Какая погода в Токио?"}]} + + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, response_obj + ) + + raw = { + call.args[1]: call.args[2] + for call in mock_safe_set_attribute.call_args_list + } + + input_raw = raw[LangfuseSpanAttributes.OBSERVATION_INPUT.value] + output_raw = raw[LangfuseSpanAttributes.OBSERVATION_OUTPUT.value] + assert "Какая погода в Токио?" in input_raw + assert "В Токио солнечно. 晴れ" in output_raw + assert "\\u" not in input_raw and "\\u" not in output_raw + assert json.loads(input_raw) == kwargs["messages"] + def test_set_langfuse_specific_attributes_with_tool_calls(self): """Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format.""" from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes diff --git a/tests/unit/litellm_core_utils/test_safe_json_dumps.py b/tests/unit/litellm_core_utils/test_safe_json_dumps.py index 1f2664f33cb..8f9233f2626 100644 --- a/tests/unit/litellm_core_utils/test_safe_json_dumps.py +++ b/tests/unit/litellm_core_utils/test_safe_json_dumps.py @@ -236,3 +236,15 @@ def test_safe_json_structure_keeps_tuples_and_drops_non_string_keys(): assert structure == {"models": ("A", "B"), "tags": ["X", "Y"], "nested": {"deep": ("C",)}} assert type(structure["models"]) is tuple assert json.loads(safe_dumps(data)) == {"models": ["a", "b"], "tags": ["x", "y"], "nested": {"deep": ["c"]}} + + +def test_ensure_ascii_default_escapes_non_ascii(): + # Default stays json.dumps-compatible: non-ASCII is escaped + assert safe_dumps({"text": "Привет"}) == '{"text": "\\u041f\\u0440\\u0438\\u0432\\u0435\\u0442"}' + + +def test_ensure_ascii_false_keeps_non_ascii(): + data = {"text": "Привет, 世界", "nested": ["ü", {"k": "é"}]} + result = safe_dumps(data, ensure_ascii=False) + assert result == '{"text": "Привет, 世界", "nested": ["ü", {"k": "é"}]}' + assert json.loads(result) == json.loads(safe_dumps(data)) From 2bbff986fdf1bdd56dbc1402392a297b21f22f04 Mon Sep 17 00:00:00 2001 From: jahngalt <26158054+jahngalt@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:09:43 +0300 Subject: [PATCH 2/2] fix(langfuse_otel): fall back to escaped JSON on unpaired surrogates; cover output branches safe_dumps(ensure_ascii=False) now checks that its result encodes as UTF-8 and otherwise returns the escaped ensure_ascii=True output, so a lone surrogate in the text cannot break OTLP export. The structure is built once. Adds raw-attribute tests for the tool_calls and output-items branches and surrogate tests for safe_dumps and the Langfuse OTEL attributes. --- litellm/litellm_core_utils/safe_json_dumps.py | 18 ++- tests/unit/integrations/test_langfuse_otel.py | 113 ++++++++++++++++++ .../test_safe_json_dumps.py | 23 ++++ 3 files changed, 148 insertions(+), 6 deletions(-) diff --git a/litellm/litellm_core_utils/safe_json_dumps.py b/litellm/litellm_core_utils/safe_json_dumps.py index 5cd892b26cc..a8227975c0f 100644 --- a/litellm/litellm_core_utils/safe_json_dumps.py +++ b/litellm/litellm_core_utils/safe_json_dumps.py @@ -93,10 +93,16 @@ def safe_dumps( """Serialize data to JSON text through safe_json_structure. ensure_ascii=False keeps non-ASCII characters as-is instead of \\uXXXX - escapes; the parsed value is identical either way. + escapes; the parsed value is identical either way. The result of an + ensure_ascii=False call is always encodable as UTF-8: if the text holds + something that is not (e.g. an unpaired surrogate such as "\\ud800"), the + escaped ensure_ascii=True output is returned instead. """ - return json.dumps( - safe_json_structure(data, max_depth, value_transform), - default=str, - ensure_ascii=ensure_ascii, - ) + structure: Final = safe_json_structure(data, max_depth, value_transform) + text: Final = json.dumps(structure, default=str, ensure_ascii=ensure_ascii) + if not ensure_ascii: + try: + text.encode("utf-8") + except UnicodeEncodeError: + return json.dumps(structure, default=str, ensure_ascii=True) + return text diff --git a/tests/unit/integrations/test_langfuse_otel.py b/tests/unit/integrations/test_langfuse_otel.py index 1ce7a85c94e..3b91853e692 100644 --- a/tests/unit/integrations/test_langfuse_otel.py +++ b/tests/unit/integrations/test_langfuse_otel.py @@ -398,6 +398,119 @@ class TestLangfuseOtelIntegration: assert "\\u" not in input_raw and "\\u" not in output_raw assert json.loads(input_raw) == kwargs["messages"] + @staticmethod + def _raw_attributes(kwargs, response_obj): + """Run the attribute setter and return the raw {attribute: value} it wrote.""" + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, response_obj + ) + return { + call.args[1]: call.args[2] + for call in mock_safe_set_attribute.call_args_list + } + + def test_set_langfuse_specific_attributes_tool_calls_keep_non_ascii_unescaped(self): + """Tool call arguments with non-ASCII text are written as-is, not as \\uXXXX escapes.""" + from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + from litellm.types.utils import ( + ChatCompletionMessageToolCall, + Choices, + Function, + ModelResponse, + ) + + response_obj = ModelResponse( + id="chatcmpl-test", + model="gpt-4o", + choices=[ + Choices( + finish_reason="tool_calls", + message={ + "role": "assistant", + "content": None, + "tool_calls": [ + ChatCompletionMessageToolCall( + function=Function( + arguments='{"greeting": "Привет, мир", "city": "東京"}', + name="say_hello", + ), + id="call_123", + type="function", + ) + ], + }, + ) + ], + ) + + raw = self._raw_attributes({}, response_obj) + + output_raw = raw[LangfuseSpanAttributes.OBSERVATION_OUTPUT.value] + assert "Привет, мир" in output_raw and "東京" in output_raw + assert "\\u" not in output_raw + assert json.loads(output_raw) == [ + { + "id": "chatcmpl-test", + "name": "say_hello", + "arguments": {"greeting": "Привет, мир", "city": "東京"}, + "call_id": "call_123", + "type": "function_call", + } + ] + + def test_set_langfuse_specific_attributes_output_items_keep_non_ascii_unescaped( + self, + ): + """Responses API output items with non-ASCII text are written as-is, not as \\uXXXX escapes.""" + from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + response_obj = ResponsesAPIResponse( + id="response-unicode", + created_at=1625247600, + output=[ + ResponseOutputMessage( + id="msg-001", + type="message", + role="assistant", + status="completed", + content=[ + ResponseOutputText( + annotations=[], + text="こんにちは, Ünïcödé", + type="output_text", + ) + ], + ) + ], + ) + + raw = self._raw_attributes({"call_type": "responses"}, response_obj) + + output_raw = raw[LangfuseSpanAttributes.OBSERVATION_OUTPUT.value] + assert "こんにちは, Ünïcödé" in output_raw + assert "\\u" not in output_raw + assert json.loads(output_raw) == [ + {"role": "assistant", "content": "こんにちは, Ünïcödé"} + ] + + def test_set_langfuse_specific_attributes_unpaired_surrogate_stays_utf8_encodable( + self, + ): + """An unpaired surrogate must not yield a string that cannot be UTF-8 encoded (OTLP export).""" + from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + + kwargs = {"messages": [{"role": "user", "content": "Привет \ud800 мир"}]} + + raw = self._raw_attributes(kwargs, None) + + input_raw = raw[LangfuseSpanAttributes.OBSERVATION_INPUT.value] + input_raw.encode("utf-8") + assert json.loads(input_raw) == kwargs["messages"] + def test_set_langfuse_specific_attributes_with_tool_calls(self): """Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format.""" from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes diff --git a/tests/unit/litellm_core_utils/test_safe_json_dumps.py b/tests/unit/litellm_core_utils/test_safe_json_dumps.py index 8f9233f2626..e20cf15be54 100644 --- a/tests/unit/litellm_core_utils/test_safe_json_dumps.py +++ b/tests/unit/litellm_core_utils/test_safe_json_dumps.py @@ -248,3 +248,26 @@ def test_ensure_ascii_false_keeps_non_ascii(): result = safe_dumps(data, ensure_ascii=False) assert result == '{"text": "Привет, 世界", "nested": ["ü", {"k": "é"}]}' assert json.loads(result) == json.loads(safe_dumps(data)) + + +def test_ensure_ascii_false_unpaired_surrogate_falls_back_to_escaped_json(): + data = {"t": "a\ud800b"} + result = safe_dumps(data, ensure_ascii=False) + result.encode("utf-8") + assert result == '{"t": "a\\ud800b"}' + assert json.loads(result) == data + + +def test_ensure_ascii_false_surrogate_fallback_escapes_all_non_ascii(): + # The fallback is the previous escaped output as a whole, not a partial repair + data = {"t": "Привет \ud800", "u": "こんにちは"} + result = safe_dumps(data, ensure_ascii=False) + result.encode("utf-8") + assert result == safe_dumps(data) + assert json.loads(result) == data + + +def test_ensure_ascii_false_valid_non_ascii_not_escaped_alongside_ascii_only_default(): + result = safe_dumps({"t": "Ünïcödé"}, ensure_ascii=False) + assert result == '{"t": "Ünïcödé"}' + result.encode("utf-8")