mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
fix(tests): follow the anthropic pass_through rename after merging main
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
9e0d6f59aa
commit
72769467f7
4 changed files with 25 additions and 17 deletions
|
|
@ -63,7 +63,7 @@
|
|||
- {id: quota_management.spend_tracking.service_tier.bills_tier_rates, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier, assertions: [bills_tier_rates], exercised_on: [chat_completions], source: "cost_calculator.py", rationale: "A priority service_tier call bills input, output, and reasoning at the deployment's *_priority rates and records the tier on the row (#35923, #35925)"}
|
||||
- {id: quota_management.spend_tracking.service_tier_stream.records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [chat_completions], source: "litellm_core_utils/streaming_chunk_builder_utils.py", fail_before_fix: proven, rationale: "A streamed call with no service_tier requested bills at the rates of the tier OpenAI stamps on its chunks and records that served tier on the row; the reassembled stream dropped the provider tier so the row recorded none and priced at the default rates"}
|
||||
- {id: quota_management.spend_tracking.service_tier_stream.responses_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [responses], source: "responses/streaming_iterator.py", rationale: "A streamed /v1/responses call bills at the tier carried on the response.completed event's inner response and records that served tier on the spend row"}
|
||||
- {id: quota_management.spend_tracking.service_tier_stream.messages_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [messages], source: "llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py", rationale: "A streamed /v1/messages call on an OpenAI-backed deployment bills at the tier OpenAI served; the Anthropic wire format has no tier field, so the spend row is the only record of it"}
|
||||
- {id: quota_management.spend_tracking.service_tier_stream.messages_records_served_tier, module: quota_management, tier: P1, behavior: spend_tracking, variant: service_tier_stream, assertions: [records_served_tier], exercised_on: [messages], source: "llms/anthropic/pass_through/adapters/streaming_iterator.py", rationale: "A streamed /v1/messages call on an OpenAI-backed deployment bills at the tier OpenAI served; the Anthropic wire format has no tier field, so the spend row is the only record of it"}
|
||||
- {id: quota_management.spend_tracking.cost_headers.additive_components, module: quota_management, tier: P1, behavior: spend_tracking, variant: cost_headers, assertions: [additive_components], exercised_on: [chat_completions], source: "proxy/common_request_processing.py", rationale: "The x-litellm-response-cost-* component headers sum to the total, input covers only fresh tokens, and reasoning stays a subset of output (#36965)"}
|
||||
- {id: quota_management.spend_tracking.passthrough_stream.injects_usage_cost, module: quota_management, tier: P1, behavior: spend_tracking, variant: passthrough_stream, assertions: [injects_usage_cost], exercised_on: [openai_passthrough], source: "proxy/pass_through_endpoints/streaming_handler.py", rationale: "With include_cost_in_streaming_usage on, the /openai passthrough's final streaming usage frame carries the proxy-computed cost (#36503). Uncovered: the flag is only settable in litellm_settings, and the shared e2e stack does not turn it on yet"}
|
||||
- {id: quota_management.spend_tracking.websearch_interception.bills_under_request_session, module: quota_management, tier: P1, behavior: spend_tracking, variant: websearch_interception, assertions: [bills_under_request_session], exercised_on: [messages], source: "integrations/websearch_interception/handler.py", fail_before_fix: proven, rationale: "A web_search server tool the proxy intercepts into litellm.asearch writes its own asearch spend row, and that row carries the parent request's session_id so the session view counts the search and its cost next to the turn that triggered it (LIT-8063)"}
|
||||
|
|
|
|||
|
|
@ -7346,10 +7346,10 @@ class TestStreamingClientDisconnectBilling:
|
|||
through the translate_completion_output_params_streaming result to the
|
||||
inner chat stream's collected chunks or a disconnect bills nothing.
|
||||
"""
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
|
||||
from litellm.llms.anthropic.pass_through.adapters.streaming_iterator import (
|
||||
AnthropicSSEStream,
|
||||
)
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
from litellm.llms.anthropic.pass_through.adapters.transformation import (
|
||||
AnthropicAdapter,
|
||||
)
|
||||
from litellm.router import FallbackAwareAnthropicMessagesStream
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ from unittest.mock import MagicMock
|
|||
|
||||
import pytest
|
||||
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
|
||||
from litellm.llms.anthropic.pass_through.adapters.streaming_iterator import (
|
||||
AnthropicSSEStream,
|
||||
AnthropicStreamWrapper,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,21 +1,21 @@
|
|||
import asyncio
|
||||
from typing import Any, AsyncIterator, Dict, List
|
||||
import datetime
|
||||
from collections.abc import AsyncIterator
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
import datetime
|
||||
|
||||
import litellm
|
||||
from litellm._internal_context import in_post_response_phase
|
||||
from litellm.caching.caching import Cache, LiteLLMCacheType
|
||||
from litellm.caching.caching_handler import LLMCachingHandler
|
||||
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
||||
from litellm.llms.anthropic.pass_through.messages import handler
|
||||
from litellm.llms.anthropic.pass_through.messages.response_cache import (
|
||||
AnthropicMessagesStreamCacheWriter,
|
||||
)
|
||||
|
||||
STREAM_EVENTS: List[bytes] = [
|
||||
STREAM_EVENTS: list[bytes] = [
|
||||
b'event: message_start\ndata: {"type": "message_start", "message": {"id": "msg_stream_1", "type": "message", '
|
||||
b'"role": "assistant", "model": "claude-sonnet-4-5", "content": [], "stop_reason": null, '
|
||||
b'"usage": {"input_tokens": 10, "output_tokens": 0}}}\n\n',
|
||||
|
|
@ -30,7 +30,7 @@ STREAM_EVENTS: List[bytes] = [
|
|||
]
|
||||
|
||||
|
||||
def _anthropic_response(message_id: str, text: str) -> Dict[str, Any]:
|
||||
def _anthropic_response(message_id: str, text: str) -> dict[str, Any]:
|
||||
return {
|
||||
"id": message_id,
|
||||
"type": "message",
|
||||
|
|
@ -45,24 +45,32 @@ def _anthropic_response(message_id: str, text: str) -> Dict[str, Any]:
|
|||
class _CountingHandler:
|
||||
"""Stands in for the provider dispatch so cache hits are observable as skipped calls."""
|
||||
|
||||
def __init__(self, results: List[Any]) -> None:
|
||||
def __init__(self, results: list[Any]) -> None:
|
||||
self.results = results
|
||||
self.calls: List[Dict[str, Any]] = []
|
||||
self.calls: list[dict[str, Any]] = []
|
||||
|
||||
def __call__(self, *args: Any, **kwargs: Any) -> Any:
|
||||
self.calls.append(kwargs)
|
||||
return self.results[min(len(self.calls) - 1, len(self.results) - 1)]
|
||||
|
||||
|
||||
async def _byte_stream(chunks: List[bytes]) -> AsyncIterator[bytes]:
|
||||
async def _byte_stream(chunks: list[bytes]) -> AsyncIterator[bytes]:
|
||||
for chunk in chunks:
|
||||
yield chunk
|
||||
|
||||
|
||||
async def _collect(stream: AsyncIterator[bytes]) -> List[bytes]:
|
||||
async def _collect(stream: AsyncIterator[bytes]) -> list[bytes]:
|
||||
return [chunk async for chunk in stream]
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
async def _drain_logging_worker():
|
||||
# anthropic_messages enqueues success logging on the global LoggingWorker; left
|
||||
# pending, those coroutines revive on the next test's loop and pollute call counts
|
||||
yield
|
||||
await GLOBAL_LOGGING_WORKER.flush()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def local_cache():
|
||||
previous_cache = litellm.cache
|
||||
|
|
@ -72,7 +80,7 @@ def local_cache():
|
|||
|
||||
|
||||
@pytest.fixture
|
||||
def request_kwargs() -> Dict[str, Any]:
|
||||
def request_kwargs() -> dict[str, Any]:
|
||||
return {
|
||||
"model": "anthropic/claude-sonnet-4-5",
|
||||
"custom_llm_provider": "anthropic",
|
||||
|
|
@ -176,8 +184,8 @@ async def test_multibyte_utf8_split_across_chunks_streams_and_caches(local_cache
|
|||
multibyte_delta = (
|
||||
'event: content_block_delta\ndata: {"type": "content_block_delta", "index": 0, '
|
||||
'"delta": {"type": "text_delta", "text": "ALPHA €"}}\n\n'
|
||||
).encode("utf-8")
|
||||
split_at = multibyte_delta.index("€".encode("utf-8")) + 1
|
||||
).encode()
|
||||
split_at = multibyte_delta.index("€".encode()) + 1
|
||||
chunks = STREAM_EVENTS[:2] + [multibyte_delta[:split_at], multibyte_delta[split_at:]] + STREAM_EVENTS[3:]
|
||||
fake_handler = _CountingHandler([_byte_stream(chunks), _byte_stream([b"event: never_used\n\n"])])
|
||||
monkeypatch.setattr(handler, "anthropic_messages_handler", fake_handler)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue