From a6d2332f7a19822ada77c4fd2cc13b853d1f9cad Mon Sep 17 00:00:00 2001 From: kerry Date: Wed, 16 Sep 2026 01:24:35 +0000 Subject: [PATCH] test(responses): drive the usage-estimate failure path without patching litellm Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../responses/test_streaming_iterator.py | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/tests/test_litellm/responses/test_streaming_iterator.py b/tests/test_litellm/responses/test_streaming_iterator.py index 7bd03863717..d3849e6ffce 100644 --- a/tests/test_litellm/responses/test_streaming_iterator.py +++ b/tests/test_litellm/responses/test_streaming_iterator.py @@ -17,6 +17,7 @@ from litellm.llms.base_llm.responses.transformation import BaseResponsesAPIConfi from litellm.responses.streaming_iterator import ( ResponsesAPIStreamingIterator, SyncResponsesAPIStreamingIterator, + _estimate_usage_from_text, ) from litellm.types.llms.openai import ( ResponseAPIUsage, @@ -796,8 +797,13 @@ async def test_completed_event_without_usage_counts_multimodal_input_as_messages @pytest.mark.asyncio async def test_completed_event_survives_a_failing_usage_estimate(): - """A raising token_counter must not break a stream that previously completed: - the estimate is best-effort and falls back to usage None.""" + """A malformed request input that makes the message transformer raise must not + break a stream that previously completed: the estimate is best-effort and + falls back to usage None.""" + malformed_input: Final = [{"type": "message", "role": "user", "content": 42}] + with pytest.raises(ValueError): + _estimate_usage_from_text("gpt-4o-mini", malformed_input, {"input": malformed_input}, "hello world") + response = _responses_api_response_without_usage() iterator = _make_iterator( sse_events=[ @@ -806,13 +812,12 @@ async def test_completed_event_survives_a_failing_usage_estimate(): ], logging_obj=_logging_obj_stub(), config=_mock_config_with_completed_response(response), - request_data={"input": "count these input tokens please"}, + request_data={"input": malformed_input}, ) - with patch.object(litellm, "token_counter", side_effect=RuntimeError("tokenizer exploded")): - yielded: list = [] - async for chunk in iterator: - yielded.append(chunk) + yielded: list = [] + async for chunk in iterator: + yielded.append(chunk) assert yielded assert iterator.completed_response.response.usage is None