From 72769467f71be4eb0809e6f6335bcaad913dc36d Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 26 Sep 2026 23:22:30 +0000 Subject: [PATCH] fix(tests): follow the anthropic pass_through rename after merging main Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../coverage_registry/quota_management.yaml | 2 +- .../proxy/test_common_request_processing.py | 4 +-- .../test_streaming_iterator_sse_stream.py | 2 +- .../messages/test_response_cache.py | 34 ++++++++++++------- 4 files changed, 25 insertions(+), 17 deletions(-) diff --git a/tests/e2e/coverage_registry/quota_management.yaml b/tests/e2e/coverage_registry/quota_management.yaml index 02eb4bd785f..163de67fc41 100644 --- a/tests/e2e/coverage_registry/quota_management.yaml +++ b/tests/e2e/coverage_registry/quota_management.yaml @@ -63,7 +63,7 @@ - {id: quota_management.spend_tracking.service_tier.bills_tier_rates, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier, assertions: [bills_tier_rates], exercised_on: [chat_completions], source: "cost_calculator.py", rationale: "A priority service_tier call bills input, output, and reasoning at the deployment's *_priority rates and records the tier on the row (#35923, #35925)"} - {id: quota_management.spend_tracking.service_tier_stream.records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [chat_completions], source: "litellm_core_utils/streaming_chunk_builder_utils.py", fail_before_fix: proven, rationale: "A streamed call with no service_tier requested bills at the rates of the tier OpenAI stamps on its chunks and records that served tier on the row; the reassembled stream dropped the provider tier so the row recorded none and priced at the default rates"} - {id: quota_management.spend_tracking.service_tier_stream.responses_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [responses], source: "responses/streaming_iterator.py", rationale: "A streamed /v1/responses call bills at the tier carried on the response.completed event's inner response and records that served tier on the spend row"} -- {id: quota_management.spend_tracking.service_tier_stream.messages_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [messages], source: "llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py", rationale: "A streamed /v1/messages call on an OpenAI-backed deployment bills at the tier OpenAI served; the Anthropic wire format has no tier field, so the spend row is the only record of it"} +- {id: quota_management.spend_tracking.service_tier_stream.messages_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [messages], source: "llms/anthropic/pass_through/adapters/streaming_iterator.py", rationale: "A streamed /v1/messages call on an OpenAI-backed deployment bills at the tier OpenAI served; the Anthropic wire format has no tier field, so the spend row is the only record of it"} - {id: quota_management.spend_tracking.cost_headers.additive_components, module: quota_management, tier: P1, behavior: spend_tracking, variant: cost_headers, assertions: [additive_components], exercised_on: [chat_completions], source: "proxy/common_request_processing.py", rationale: "The x-litellm-response-cost-* component headers sum to the total, input covers only fresh tokens, and reasoning stays a subset of output (#36965)"} - {id: quota_management.spend_tracking.passthrough_stream.injects_usage_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: passthrough_stream, assertions: [injects_usage_cost], exercised_on: [openai_passthrough], source: "proxy/pass_through_endpoints/streaming_handler.py", rationale: "With include_cost_in_streaming_usage on, the /openai passthrough's final streaming usage frame carries the proxy-computed cost (#36503). Uncovered: the flag is only settable in litellm_settings, and the shared e2e stack does not turn it on yet"} - {id: quota_management.spend_tracking.websearch_interception.bills_under_request_session, module: quota_management, tier: P1, behavior: spend_tracking, variant: websearch_interception, assertions: [bills_under_request_session], exercised_on: [messages], source: "integrations/websearch_interception/handler.py", fail_before_fix: proven, rationale: "A web_search server tool the proxy intercepts into litellm.asearch writes its own asearch spend row, and that row carries the parent request's session_id so the session view counts the search and its cost next to the turn that triggered it (LIT-8063)"} diff --git a/tests/test_litellm/proxy/test_common_request_processing.py b/tests/test_litellm/proxy/test_common_request_processing.py index 4b54df98aeb..8485c286a30 100644 --- a/tests/test_litellm/proxy/test_common_request_processing.py +++ b/tests/test_litellm/proxy/test_common_request_processing.py @@ -7346,10 +7346,10 @@ class TestStreamingClientDisconnectBilling: through the translate_completion_output_params_streaming result to the inner chat stream's collected chunks or a disconnect bills nothing. """ - from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import ( + from litellm.llms.anthropic.pass_through.adapters.streaming_iterator import ( AnthropicSSEStream, ) - from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + from litellm.llms.anthropic.pass_through.adapters.transformation import ( AnthropicAdapter, ) from litellm.router import FallbackAwareAnthropicMessagesStream diff --git a/tests/unit/llms/anthropic/pass_through/adapters/test_streaming_iterator_sse_stream.py b/tests/unit/llms/anthropic/pass_through/adapters/test_streaming_iterator_sse_stream.py index 5fbf9f8be99..fbbbc579d94 100644 --- a/tests/unit/llms/anthropic/pass_through/adapters/test_streaming_iterator_sse_stream.py +++ b/tests/unit/llms/anthropic/pass_through/adapters/test_streaming_iterator_sse_stream.py @@ -10,7 +10,7 @@ from unittest.mock import MagicMock import pytest -from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import ( +from litellm.llms.anthropic.pass_through.adapters.streaming_iterator import ( AnthropicSSEStream, AnthropicStreamWrapper, ) diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_response_cache.py b/tests/unit/llms/anthropic/pass_through/messages/test_response_cache.py index c172c465632..9984ba01d03 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_response_cache.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_response_cache.py @@ -1,21 +1,21 @@ import asyncio -from typing import Any, AsyncIterator, Dict, List +import datetime +from collections.abc import AsyncIterator +from typing import Any import pytest - -import datetime - import litellm from litellm._internal_context import in_post_response_phase from litellm.caching.caching import Cache, LiteLLMCacheType from litellm.caching.caching_handler import LLMCachingHandler +from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER from litellm.llms.anthropic.pass_through.messages import handler from litellm.llms.anthropic.pass_through.messages.response_cache import ( AnthropicMessagesStreamCacheWriter, ) -STREAM_EVENTS: List[bytes] = [ +STREAM_EVENTS: list[bytes] = [ b'event: message_start\ndata: {"type": "message_start", "message": {"id": "msg_stream_1", "type": "message", ' b'"role": "assistant", "model": "claude-sonnet-4-5", "content": [], "stop_reason": null, ' b'"usage": {"input_tokens": 10, "output_tokens": 0}}}\n\n', @@ -30,7 +30,7 @@ STREAM_EVENTS: List[bytes] = [ ] -def _anthropic_response(message_id: str, text: str) -> Dict[str, Any]: +def _anthropic_response(message_id: str, text: str) -> dict[str, Any]: return { "id": message_id, "type": "message", @@ -45,24 +45,32 @@ def _anthropic_response(message_id: str, text: str) -> Dict[str, Any]: class _CountingHandler: """Stands in for the provider dispatch so cache hits are observable as skipped calls.""" - def __init__(self, results: List[Any]) -> None: + def __init__(self, results: list[Any]) -> None: self.results = results - self.calls: List[Dict[str, Any]] = [] + self.calls: list[dict[str, Any]] = [] def __call__(self, *args: Any, **kwargs: Any) -> Any: self.calls.append(kwargs) return self.results[min(len(self.calls) - 1, len(self.results) - 1)] -async def _byte_stream(chunks: List[bytes]) -> AsyncIterator[bytes]: +async def _byte_stream(chunks: list[bytes]) -> AsyncIterator[bytes]: for chunk in chunks: yield chunk -async def _collect(stream: AsyncIterator[bytes]) -> List[bytes]: +async def _collect(stream: AsyncIterator[bytes]) -> list[bytes]: return [chunk async for chunk in stream] +@pytest.fixture(autouse=True) +async def _drain_logging_worker(): + # anthropic_messages enqueues success logging on the global LoggingWorker; left + # pending, those coroutines revive on the next test's loop and pollute call counts + yield + await GLOBAL_LOGGING_WORKER.flush() + + @pytest.fixture def local_cache(): previous_cache = litellm.cache @@ -72,7 +80,7 @@ def local_cache(): @pytest.fixture -def request_kwargs() -> Dict[str, Any]: +def request_kwargs() -> dict[str, Any]: return { "model": "anthropic/claude-sonnet-4-5", "custom_llm_provider": "anthropic", @@ -176,8 +184,8 @@ async def test_multibyte_utf8_split_across_chunks_streams_and_caches(local_cache multibyte_delta = ( 'event: content_block_delta\ndata: {"type": "content_block_delta", "index": 0, ' '"delta": {"type": "text_delta", "text": "ALPHA €"}}\n\n' - ).encode("utf-8") - split_at = multibyte_delta.index("€".encode("utf-8")) + 1 + ).encode() + split_at = multibyte_delta.index("€".encode()) + 1 chunks = STREAM_EVENTS[:2] + [multibyte_delta[:split_at], multibyte_delta[split_at:]] + STREAM_EVENTS[3:] fake_handler = _CountingHandler([_byte_stream(chunks), _byte_stream([b"event: never_used\n\n"])]) monkeypatch.setattr(handler, "anthropic_messages_handler", fake_handler)