fix(tests): follow the anthropic pass_through rename after merging main

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-26 23:22:30 +00:00
parent 9e0d6f59aa
commit 72769467f7
4 changed files with 25 additions and 17 deletions

View file

@ -63,7 +63,7 @@
- {id: quota_management.spend_tracking.service_tier.bills_tier_rates, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier, assertions: [bills_tier_rates], exercised_on: [chat_completions], source: "cost_calculator.py", rationale: "A priority service_tier call bills input, output, and reasoning at the deployment's *_priority rates and records the tier on the row (#35923, #35925)"}
- {id: quota_management.spend_tracking.service_tier_stream.records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [chat_completions], source: "litellm_core_utils/streaming_chunk_builder_utils.py", fail_before_fix: proven, rationale: "A streamed call with no service_tier requested bills at the rates of the tier OpenAI stamps on its chunks and records that served tier on the row; the reassembled stream dropped the provider tier so the row recorded none and priced at the default rates"}
- {id: quota_management.spend_tracking.service_tier_stream.responses_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [responses], source: "responses/streaming_iterator.py", rationale: "A streamed /v1/responses call bills at the tier carried on the response.completed event's inner response and records that served tier on the spend row"}
- {id: quota_management.spend_tracking.service_tier_stream.messages_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [messages], source: "llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py", rationale: "A streamed /v1/messages call on an OpenAI-backed deployment bills at the tier OpenAI served; the Anthropic wire format has no tier field, so the spend row is the only record of it"}
- {id: quota_management.spend_tracking.service_tier_stream.messages_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [messages], source: "llms/anthropic/pass_through/adapters/streaming_iterator.py", rationale: "A streamed /v1/messages call on an OpenAI-backed deployment bills at the tier OpenAI served; the Anthropic wire format has no tier field, so the spend row is the only record of it"}
- {id: quota_management.spend_tracking.cost_headers.additive_components, module: quota_management, tier: P1, behavior: spend_tracking, variant: cost_headers, assertions: [additive_components], exercised_on: [chat_completions], source: "proxy/common_request_processing.py", rationale: "The x-litellm-response-cost-* component headers sum to the total, input covers only fresh tokens, and reasoning stays a subset of output (#36965)"}
- {id: quota_management.spend_tracking.passthrough_stream.injects_usage_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: passthrough_stream, assertions: [injects_usage_cost], exercised_on: [openai_passthrough], source: "proxy/pass_through_endpoints/streaming_handler.py", rationale: "With include_cost_in_streaming_usage on, the /openai passthrough's final streaming usage frame carries the proxy-computed cost (#36503). Uncovered: the flag is only settable in litellm_settings, and the shared e2e stack does not turn it on yet"}
- {id: quota_management.spend_tracking.websearch_interception.bills_under_request_session, module: quota_management, tier: P1, behavior: spend_tracking, variant: websearch_interception, assertions: [bills_under_request_session], exercised_on: [messages], source: "integrations/websearch_interception/handler.py", fail_before_fix: proven, rationale: "A web_search server tool the proxy intercepts into litellm.asearch writes its own asearch spend row, and that row carries the parent request's session_id so the session view counts the search and its cost next to the turn that triggered it (LIT-8063)"}

View file

@ -7346,10 +7346,10 @@ class TestStreamingClientDisconnectBilling:
through the translate_completion_output_params_streaming result to the
inner chat stream's collected chunks or a disconnect bills nothing.
"""
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
from litellm.llms.anthropic.pass_through.adapters.streaming_iterator import (
AnthropicSSEStream,
)
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
from litellm.llms.anthropic.pass_through.adapters.transformation import (
AnthropicAdapter,
)
from litellm.router import FallbackAwareAnthropicMessagesStream

View file

@ -10,7 +10,7 @@ from unittest.mock import MagicMock
import pytest
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
from litellm.llms.anthropic.pass_through.adapters.streaming_iterator import (
AnthropicSSEStream,
AnthropicStreamWrapper,
)

View file

@ -1,21 +1,21 @@
import asyncio
from typing import Any, AsyncIterator, Dict, List
import datetime
from collections.abc import AsyncIterator
from typing import Any
import pytest
import datetime
import litellm
from litellm._internal_context import in_post_response_phase
from litellm.caching.caching import Cache, LiteLLMCacheType
from litellm.caching.caching_handler import LLMCachingHandler
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
from litellm.llms.anthropic.pass_through.messages import handler
from litellm.llms.anthropic.pass_through.messages.response_cache import (
AnthropicMessagesStreamCacheWriter,
)
STREAM_EVENTS: List[bytes] = [
STREAM_EVENTS: list[bytes] = [
b'event: message_start\ndata: {"type": "message_start", "message": {"id": "msg_stream_1", "type": "message", '
b'"role": "assistant", "model": "claude-sonnet-4-5", "content": [], "stop_reason": null, '
b'"usage": {"input_tokens": 10, "output_tokens": 0}}}\n\n',
@ -30,7 +30,7 @@ STREAM_EVENTS: List[bytes] = [
]
def _anthropic_response(message_id: str, text: str) -> Dict[str, Any]:
def _anthropic_response(message_id: str, text: str) -> dict[str, Any]:
return {
"id": message_id,
"type": "message",
@ -45,24 +45,32 @@ def _anthropic_response(message_id: str, text: str) -> Dict[str, Any]:
class _CountingHandler:
"""Stands in for the provider dispatch so cache hits are observable as skipped calls."""
def __init__(self, results: List[Any]) -> None:
def __init__(self, results: list[Any]) -> None:
self.results = results
self.calls: List[Dict[str, Any]] = []
self.calls: list[dict[str, Any]] = []
def __call__(self, *args: Any, **kwargs: Any) -> Any:
self.calls.append(kwargs)
return self.results[min(len(self.calls) - 1, len(self.results) - 1)]
async def _byte_stream(chunks: List[bytes]) -> AsyncIterator[bytes]:
async def _byte_stream(chunks: list[bytes]) -> AsyncIterator[bytes]:
for chunk in chunks:
yield chunk
async def _collect(stream: AsyncIterator[bytes]) -> List[bytes]:
async def _collect(stream: AsyncIterator[bytes]) -> list[bytes]:
return [chunk async for chunk in stream]
@pytest.fixture(autouse=True)
async def _drain_logging_worker():
# anthropic_messages enqueues success logging on the global LoggingWorker; left
# pending, those coroutines revive on the next test's loop and pollute call counts
yield
await GLOBAL_LOGGING_WORKER.flush()
@pytest.fixture
def local_cache():
previous_cache = litellm.cache
@ -72,7 +80,7 @@ def local_cache():
@pytest.fixture
def request_kwargs() -> Dict[str, Any]:
def request_kwargs() -> dict[str, Any]:
return {
"model": "anthropic/claude-sonnet-4-5",
"custom_llm_provider": "anthropic",
@ -176,8 +184,8 @@ async def test_multibyte_utf8_split_across_chunks_streams_and_caches(local_cache
multibyte_delta = (
'event: content_block_delta\ndata: {"type": "content_block_delta", "index": 0, '
'"delta": {"type": "text_delta", "text": "ALPHA €"}}\n\n'
).encode("utf-8")
split_at = multibyte_delta.index("€".encode("utf-8")) + 1
).encode()
split_at = multibyte_delta.index("€".encode()) + 1
chunks = STREAM_EVENTS[:2] + [multibyte_delta[:split_at], multibyte_delta[split_at:]] + STREAM_EVENTS[3:]
fake_handler = _CountingHandler([_byte_stream(chunks), _byte_stream([b"event: never_used\n\n"])])
monkeypatch.setattr(handler, "anthropic_messages_handler", fake_handler)