diff --git a/litellm/litellm_core_utils/logging_utils.py b/litellm/litellm_core_utils/logging_utils.py index 5e247324cea..5ac6b5435dc 100644 --- a/litellm/litellm_core_utils/logging_utils.py +++ b/litellm/litellm_core_utils/logging_utils.py @@ -43,7 +43,7 @@ Helper utils used for logging callbacks # Regex matching data-URI base64 content: "data:;base64," # Captures: group(1)=mime_type, group(2)=base64_payload -_DATA_URI_RE: Final = re.compile(r"data:([^;]+);base64,([A-Za-z0-9+/=]+)") +_DATA_URI_RE: Final = re.compile(r"data:([^;,\s]{1,255});base64,([A-Za-z0-9+/=]+)") # Maximum nesting depth for _truncate_base64_in_value to guard against # pathological payloads. OpenAI message format is typically 3-4 levels deep. diff --git a/tests/unit/litellm_core_utils/test_logging_utils.py b/tests/unit/litellm_core_utils/test_logging_utils.py index 7b8db097d47..edf0dc7960b 100644 --- a/tests/unit/litellm_core_utils/test_logging_utils.py +++ b/tests/unit/litellm_core_utils/test_logging_utils.py @@ -95,6 +95,18 @@ class TestTruncateBase64InString: result = _truncate_base64_in_string(text) assert result.count("base64_data truncated") == 2 + @pytest.mark.timeout(10) + @pytest.mark.parametrize( + "text", + [ + 'data: {"choices": [{"delta": {"content": "hi"}}]}\n\n' * 50_000, + "data:" * 200_000, + ], + ids=["sse_lines", "whitespace_free_prefixes"], + ) + def test_repeated_data_prefixes_without_data_uris_are_scanned_in_linear_time(self, text: str): + assert _truncate_base64_in_string(text) == text + def test_no_data_uri(self): text = "hello world, no base64 here" assert _truncate_base64_in_string(text) == text