From 2fead4f6aa0064434ac649c0ea8a9f22686e1be3 Mon Sep 17 00:00:00 2001 From: jibanez-staticduo Date: Thu, 1 Oct 2026 04:39:11 +0200 Subject: [PATCH] fix(live): filter terminal response content in realtime logs --- .../litellm_core_utils/realtime_streaming.py | 24 +++++++++++- .../test_realtime_streaming.py | 38 +++++++++++++++++++ 2 files changed, 61 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/realtime_streaming.py b/litellm/litellm_core_utils/realtime_streaming.py index 1713bc11102..98dd7abf903 100644 --- a/litellm/litellm_core_utils/realtime_streaming.py +++ b/litellm/litellm_core_utils/realtime_streaming.py @@ -283,7 +283,29 @@ class RealTimeStreaming: if message_obj.get("type") == "response.event" and isinstance(message_obj.get("event"), dict): nested: Final = message_obj["event"] if nested.get("type") in ("response.completed", "response.incomplete", "response.failed"): - self.messages.append(TypeAdapter(OpenAILiveResponseEvent).validate_python(message_obj)) + response: Final = nested.get("response") + # Retain billing evidence even when response content is excluded from logging. + stored: Final = ( + message_obj + if self._should_store_message(message_obj) + else { + "type": "response.event", + "event": { + "type": nested["type"], + "response": { + **{ + key: value + for key, value in response.items() + if key in ("id", "created_at", "model", "usage", "service_tier") + }, + "output": [], + } + if isinstance(response, Mapping) + else None, + }, + } + ) + self.messages.append(TypeAdapter(OpenAILiveResponseEvent).validate_python(stored)) return if not self._should_store_message(message_obj): return diff --git a/tests/unit/litellm_core_utils/test_realtime_streaming.py b/tests/unit/litellm_core_utils/test_realtime_streaming.py index d74f3fcf89b..fbd3639eac0 100644 --- a/tests/unit/litellm_core_utils/test_realtime_streaming.py +++ b/tests/unit/litellm_core_utils/test_realtime_streaming.py @@ -3640,6 +3640,7 @@ def test_public_live_accounting_survives_filtered_logging(monkeypatch): "response": { "id": "resp_one", "model": "gpt-backend", + "output": [], "usage": {"total_tokens": 12}, }, }, @@ -3654,6 +3655,43 @@ def test_public_live_accounting_survives_filtered_logging(monkeypatch): assert stream.messages == events +@pytest.mark.parametrize("terminal", ["response.completed", "response.incomplete", "response.failed"]) +@pytest.mark.parametrize("allowed", [[], ["response.event"], "*"]) +def test_live_terminal_logging_filters_content_and_preserves_accounting( + monkeypatch: pytest.MonkeyPatch, terminal: str, allowed: list[str] | str +) -> None: + from litellm.cost_calculator import _live_backend_responses + + monkeypatch.setattr(litellm, "logged_real_time_event_types", allowed) + stream = RealTimeStreaming(MagicMock(), MagicMock(), MagicMock()) + response = { + "id": "resp_private", + "created_at": 1, + "model": "gpt-backend", + "output": [{"type": "message", "content": [{"type": "output_text", "text": "private answer"}]}], + "instructions": "private instructions", + "metadata": {"private": "metadata"}, + "usage": {"input_tokens": 20, "output_tokens": 10, "total_tokens": 30}, + } + event = {"type": "response.event", "event": {"type": terminal, "response": response}} + stream.store_message(event) + + stored = stream.messages[0]["event"]["response"] + if allowed: + assert stored == response + else: + assert stored == { + key: value for key, value in response.items() if key not in ("output", "instructions", "metadata") + } | {"output": []} + measured = _live_backend_responses(stream.messages) + assert len(measured) == 1 + assert measured[0].id == "resp_private" + assert measured[0].model == "gpt-backend" + assert measured[0].usage.total_tokens == 30 + assert response["instructions"] == "private instructions" + assert response["output"][0]["content"][0]["text"] == "private answer" + + @pytest.mark.parametrize("account_usage,expected", [(True, 1), (False, 0)]) def test_live_initialization_is_retained_only_by_accounting_owner(account_usage, expected): stream = RealTimeStreaming(