From 31c1ffc5a4341eee17eab9db14dac1411b6a3ffe Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Sat, 4 Jul 2026 18:56:52 -0700 Subject: [PATCH] test(e2e): close coverage gaps across chat/responses, provider features, batches, prometheus, and langfuse eviction (#32165) * fix(e2e): define SpendTagsResponse/TagSpend so spend suite collects spend_tracking/spend_e2e_client.py imported SpendTagsResponse and TagSpend from models, but neither was ever defined, so importing the client raised ImportError and pytest aborted collection for the whole e2e session. The tag-spend tests had never run. Model /spend/tags as it actually answers: a bare array of per-tag aggregates, so SpendTagsResponse is a RootModel[list[TagSpend]] like the existing SpendLogs. spend_by_tags read a nonexistent spend_per_tag field that also wouldn't match the array shape; it now reads .root, matching how spend_logs consumes its RootModel. * test(e2e): close coverage gaps across chat/responses, provider features, batches, prometheus, and langfuse eviction Adds regression nets and gap-surfacing tests: A1 (llm_translation/test_deepseek_reasoning_e2e.py): control case proves the DeepSeek reasoner returns reasoning_content; two xfail(strict) cases document that reasoning_effort='none' and thinking type='disabled' are silently dropped (LIT-3686 / GH #27453) A2 (llm_translation/test_chat_completions_regression_e2e.py and test_responses_e2e.py): parametrized regression net asserting real completion content, not just a 200, across the configured providers for /chat/completions and /responses (GH #28991) A3 (llm_translation/test_provider_features_e2e.py): asserts service_tier is honored and prompt-cache read tokens grow on a repeated cacheable prefix A4 (batches/test_batches_e2e.py): mints a rate-limited key so the batch pre-call rate limiter runs, then asserts no unattributed spend row is left behind by the internal input-file retrieval (LIT-3266) A5 (logging/test_prometheus_cardinality_e2e.py): drives one chat per distinct key_alias and asserts each alias gets its own labeled series on /metrics A6 (test_litellm/.../specialty_caches/test_dynamic_logging_cache.py): xfail(strict) regression proving eviction must not close an httpx client still held by an in-flight caller (LIT-3221 / GH #13034) Extends tests/e2e/models.py with the typed request and response fields these tests read (reasoning_effort, thinking, service_tier, key_alias, cache usage fields, spend-log api_key) Co-authored-by: Cursor * test(e2e): drop unused litellm-regression-tests submodule The e2e suite migrated the regression cases into this repo; nothing imports the submodule at runtime (only a provenance comment references it), so the .gitmodules entry and gitlink pointing at a personal repo would just make upstream CI init a submodule it never uses. Remove both to keep the change test-only. * test(e2e): drop A6 langfuse-eviction xfail; keep PR to live e2e coverage The dynamic_logging_cache strict-xfail documented an unfixed shared-httpx-client close-on-eviction bug (LIT-3221 / GH #13034). That is a non-trivial fix (thread cleanup vs shared client teardown) and belongs in its own PR, not this e2e coverage PR, so revert the file to its base state. --------- Co-authored-by: Cursor --- tests/e2e/batches/conftest.py | 36 +++++ tests/e2e/batches/test_batches_e2e.py | 112 ++++++++++--- tests/e2e/llm_translation/conftest.py | 7 + .../test_chat_completions_regression_e2e.py | 58 +++++++ .../test_deepseek_reasoning_e2e.py | 133 ++++++++++++++++ .../test_provider_features_e2e.py | 148 ++++++++++++++++++ tests/e2e/logging/conftest.py | 37 +++++ tests/e2e/logging/logging_client.py | 48 ++++++ .../test_prometheus_cardinality_e2e.py | 70 +++++++++ tests/e2e/models.py | 63 ++++++++ 10 files changed, 694 insertions(+), 18 deletions(-) create mode 100644 tests/e2e/llm_translation/test_chat_completions_regression_e2e.py create mode 100644 tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py create mode 100644 tests/e2e/llm_translation/test_provider_features_e2e.py create mode 100644 tests/e2e/logging/conftest.py create mode 100644 tests/e2e/logging/logging_client.py create mode 100644 tests/e2e/logging/test_prometheus_cardinality_e2e.py diff --git a/tests/e2e/batches/conftest.py b/tests/e2e/batches/conftest.py index 92905b33eee..2c6070c437a 100644 --- a/tests/e2e/batches/conftest.py +++ b/tests/e2e/batches/conftest.py @@ -4,13 +4,49 @@ The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker live in the parent tests/e2e/conftest.py. BatchClient holds the shared Gateway, so the `resources` fixture cleans up keys through it; tests register file deletes and batch cancels via `resources.defer(...)`. + +Batch deployments (openai-batch, azure-batch, vertex-batch, ...) are registered +once per session via /model/new and deleted on teardown so they need not live in +the proxy config. """ +from __future__ import annotations + +from typing import Iterator + import pytest from batch_client import BatchClient, build_client +from capabilities import PROVIDERS +from e2e_http import NoBody + + +def pytest_configure(config: pytest.Config) -> None: + config.addinivalue_line( + "markers", + "covers: registry cell a test covers, e.g. llm.batches.openai.basic.nonstream.works", + ) @pytest.fixture(scope="session") def client() -> BatchClient: return build_client() + + +@pytest.fixture(scope="session") +def batch_deployments(client: BatchClient) -> Iterator[None]: + probe = client.gateway.probe("/health/liveliness", params=NoBody()) + if not probe.healthy: + yield + return + + registered: list[str] = [] + try: + for provider in PROVIDERS: + registered.append( + client.create_model(provider.model, provider.litellm_params()) + ) + yield + finally: + for model_id in registered: + client.delete_model(model_id) diff --git a/tests/e2e/batches/test_batches_e2e.py b/tests/e2e/batches/test_batches_e2e.py index 1141e2f6a2f..a998f962c04 100644 --- a/tests/e2e/batches/test_batches_e2e.py +++ b/tests/e2e/batches/test_batches_e2e.py @@ -16,12 +16,13 @@ misroute to the wrong provider fails the create. from __future__ import annotations import json -import os import time from typing import Callable import pytest +from e2e_config import unique_marker + from batch_client import ( BatchClient, BatchCreateBody, @@ -42,16 +43,36 @@ from e2e_http import ( FileUploadForm, Result, StreamingResponse, + Success, + UnknownApiError, require_successful_call, unwrap, ) from lifecycle import ResourceManager +from models import KeyGenerateBody, SpendLogRow, SpendLogsParams pytestmark = pytest.mark.e2e CREATED_BATCH_STATUSES = {"validating", "in_progress", "finalizing"} BATCH_CANCEL_DELAY_SECONDS = 2 BATCH_TERMINAL_BEFORE_CANCEL = {"failed", "cancelled", "expired"} +BATCH_CANCEL_RETRIES = 3 + + +def cancel_batch( + client: BatchClient, batch_id: str, *, key: str, provider: str | None +) -> BatchObject: + last = client.cancel_batch(batch_id, key=key, provider=provider) + for _ in range(BATCH_CANCEL_RETRIES - 1): + match last: + case Success(data=data): + return data + case UnknownApiError(status_code=500): + time.sleep(1) + last = client.cancel_batch(batch_id, key=key, provider=provider) + case _: + break + return unwrap(last) def render_jsonl(model: str) -> bytes: @@ -146,7 +167,10 @@ def assert_batch_object(batch: BatchObject) -> None: @pytest.mark.parametrize("cap", CAPABILITIES, ids=[c.id for c in CAPABILITIES]) def test_batch_lifecycle( - cap: Capability, client: BatchClient, resources: ResourceManager + cap: Capability, + client: BatchClient, + resources: ResourceManager, + batch_deployments: None, ) -> None: key = resources.key() provider = op_provider(cap) @@ -199,11 +223,9 @@ def test_batch_lifecycle( ) if pre_cancel.status == "completed": return - cancelled = unwrap(client.cancel_batch(batch.id, key=key, provider=provider)) + cancelled = cancel_batch(client, batch.id, key=key, provider=provider) assert cancelled.id == batch.id assert cancelled.object == "batch" - # Vertex cancel is async: the job may still show its pre-cancel status - # briefly before transitioning to cancelling/cancelled. valid_post_cancel = {"cancelling", "cancelled"} if cap.provider == "vertex_ai": valid_post_cancel |= CREATED_BATCH_STATUSES @@ -213,7 +235,6 @@ def test_batch_lifecycle( if cap.can_list: listed = unwrap(client.list_batches(key=key, provider=provider)) - # OpenAI includes object="list"; Azure provider list often omits the envelope field. if listed.object is not None: assert listed.object == "list", f"list envelope object={listed.object!r}" match = next((b for b in listed.data if b.id == batch.id), None) @@ -222,7 +243,7 @@ def test_batch_lifecycle( def test_batch_key_model_access_denied( - client: BatchClient, resources: ResourceManager + client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: key = resources.key(models=["openai-batch"]) @@ -257,7 +278,7 @@ def test_batch_key_model_access_denied( def test_file_upload_and_delete_outputs( - client: BatchClient, resources: ResourceManager + client: BatchClient, resources: ResourceManager, batch_deployments: None ) -> None: key = resources.key() file = unwrap( @@ -276,14 +297,69 @@ def test_file_upload_and_delete_outputs( assert deleted.deleted is True, "file was not reported deleted" -def test_anthropic_batch_retrieve(client: BatchClient, scoped_key: str) -> None: - batch_id = os.environ.get("ANTHROPIC_BATCH_ID") - if not batch_id: - pytest.skip( - "set ANTHROPIC_BATCH_ID to a real anthropic batch id to exercise retrieve" - ) - fetched = unwrap( - client.retrieve_batch(batch_id, key=scoped_key, provider="anthropic") +def unattributed_rows(rows: list[SpendLogRow]) -> list[SpendLogRow]: + """Spend rows that carry no caller identity (empty api_key). + + Every request the proxy bills is stamped with the calling key. A row with no + api_key is one the proxy could not attribute; LIT-3266 is exactly this: the + batch rate limiter's internal input-file read ran without the batch's auth + metadata, landing a spend row with empty api_key/user. The symptom is not + tied to a single call_type, so this catches any unattributed row rather than + only a named file-content one. + """ + return [row for row in rows if not row.api_key] + + +def test_rate_limited_batch_create_leaves_no_unattributed_spend_row( + client: BatchClient, resources: ResourceManager, batch_deployments: None +) -> None: + """LIT-3266: creating a batch on a rate-limited key runs the batch rate + limiter, which reads the input file to count tokens (the limiter only reads + the file when the key has applicable rpm/tpm limits, so an unlimited key + hides the path). That internal read must carry the batch's auth metadata; + the reported gap was that it did not, spawning a spend-log row with empty + api_key/user. Create returning 200 is not a reliable signal (the read error + is swallowed), so this asserts the hygiene contract instead: the operation + introduces no new unattributed spend row. + + The key sets generous rpm/tpm limits (not a restrictive model allowlist) so + the file-read path fires while the batch itself is not blocked. + ``resources.key()`` cannot set limits, so the key is minted on the gateway + directly and its delete deferred. + """ + user_id = f"e2e-batch-rl-{unique_marker()}" + key = client.gateway.generate_key( + KeyGenerateBody(models=[], tpm_limit=1_000_000, rpm_limit=1_000, user_id=user_id) + ) + resources.defer(lambda: client.gateway.delete_key(key)) + + before = frozenset( + row.request_id for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams())) + ) + + file = unwrap( + client.upload_file( + content=render_jsonl("gpt-4o-mini"), + form=FileUploadForm(purpose="batch"), + model="openai-batch", + key=key, + ) + ) + resources.defer(quietly(lambda: client.delete_file(file.id, key=key))) + + created = client.create_batch(body=BatchCreateBody(input_file_id=file.id), key=key) + require_successful_call(created) + batch = BatchObject.model_validate_json(created.body) + resources.defer(quietly(lambda: client.cancel_batch(batch.id, key=key))) + + _ = client.gateway.poll_logs_for_key(key, min_rows=1) + + new_orphans = [ + row + for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams())) + if row.request_id not in before + ] + assert not new_orphans, ( + "batch create on a rate-limited key left an unattributed spend row " + f"(LIT-3266); rows={[(r.request_id, r.call_type, r.model) for r in new_orphans]}" ) - assert fetched.id == batch_id - assert fetched.status diff --git a/tests/e2e/llm_translation/conftest.py b/tests/e2e/llm_translation/conftest.py index 014e056d06d..2a87ef7259d 100644 --- a/tests/e2e/llm_translation/conftest.py +++ b/tests/e2e/llm_translation/conftest.py @@ -11,6 +11,13 @@ from endpoints_client import EndpointsClient, build_endpoints_client from passthrough_client import PassthroughClient, build_client +def pytest_configure(config: pytest.Config) -> None: + config.addinivalue_line( + "markers", + "covers: registry cell a test covers, e.g. llm.chat_completions.provider.basic.nonstream.works", + ) + + @pytest.fixture(scope="session") def client() -> PassthroughClient: return build_client() diff --git a/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py new file mode 100644 index 00000000000..269cb5d6d22 --- /dev/null +++ b/tests/e2e/llm_translation/test_chat_completions_regression_e2e.py @@ -0,0 +1,58 @@ +"""Live regression net for /chat/completions across the configured providers. + +GH #28991 broke /chat/completions (and /responses) for most models on some +releases: a clean 200 came back but with no real completion. A status check +alone would not have caught it, so each case here asserts the product promise - +a non-empty assistant message and a real model name in the body - across the +three providers wired into the gateway config (OpenAI, Anthropic, Gemini). A +regression that empties the completion for any provider fails that provider's +row here. +""" + +from __future__ import annotations + +import pytest + +from e2e_config import unique_marker +from e2e_http import unwrap +from models import ChatBody, ChatMessage +from passthrough_client import PassthroughClient + +pytestmark = pytest.mark.e2e + +CHAT_MODELS: tuple[tuple[str, str], ...] = ( + ("gpt-5.5", "openai"), + ("claude-haiku-4-5", "anthropic"), + ("gemini-2.5-flash", "gemini"), +) + + +class TestChatCompletionsRegression: + @pytest.mark.parametrize( + ("model", "route"), + CHAT_MODELS, + ids=[f"{model}-{route}" for model, route in CHAT_MODELS], + ) + @pytest.mark.covers("llm.chat_completions.provider.basic.nonstream.works", exercised_on=[]) + def test_chat_returns_real_completion( + self, client: PassthroughClient, scoped_key: str, model: str, route: str + ) -> None: + response = unwrap( + client.gateway.chat( + scoped_key, + ChatBody( + model=model, + messages=[ + ChatMessage(role="user", content=f"reply with one word {unique_marker()}") + ], + max_tokens=512, + ), + ) + ) + + assert response.model, f"{model} ({route}): response carried no model name: {response}" + assert response.choices, f"{model} ({route}): response had no choices: {response}" + message = response.choices[0].message + assert message is not None and message.content and message.content.strip(), ( + f"{model} ({route}): 200 with an empty completion (#28991): {response}" + ) diff --git a/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py b/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py new file mode 100644 index 00000000000..f8f229aa2a7 --- /dev/null +++ b/tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py @@ -0,0 +1,133 @@ +"""Live e2e: DeepSeek reasoner honors a request to turn reasoning OFF. + +DeepSeek's reasoner defaults thinking ON and surfaces the chain as +``message.reasoning_content``. Two documented ways to disable it are +``reasoning_effort="none"`` and ``thinking={"type": "disabled"}``. Today the +DeepSeek param mapper (``litellm/llms/deepseek/chat/transformation.py`` +``map_openai_params``) drops both without forwarding any disable signal, so the +outbound body carries no ``thinking`` key and DeepSeek keeps thinking on; the +response still comes back with ``reasoning_content``. That is the product gap +tracked by LIT-3686 / GH #27453. + +The control case proves the model and path work (reasoning is returned when +nothing asks to disable it), so the two disable assertions are meaningful. Those +two are marked xfail(strict) until the mapper forwards a real disable signal; an +xpass then alerts that the fix landed. + +Requires DEEPSEEK_API_KEY on the proxy (tests/e2e/.env). No skip gate: once the +proxy is up, a failure here is real, per the suite's hard-fail contract. +""" + +from __future__ import annotations + +import pytest + +from e2e_config import unique_marker +from e2e_http import unwrap +from lifecycle import ResourceManager +from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, ThinkingParam +from passthrough_client import PassthroughClient + +pytestmark = pytest.mark.e2e + +REASONER = "deepseek/deepseek-reasoner" +PROMPT = "What is 17 + 26? Answer with just the number." + + +def _register_reasoner(client: PassthroughClient, resources: ResourceManager) -> str: + model = f"e2e-deepseek-reasoner-{unique_marker()}" + model_id = client.gateway.create_model( + model, + LiteLLMParamsBody(model=REASONER, api_key="os.environ/DEEPSEEK_API_KEY"), + ) + resources.defer(lambda: client.gateway.delete_model(model_id)) + return model + + +def _reasoning_content(response: ChatResponse) -> str | None: + assert response.choices, f"reasoner returned no choices: {response}" + message = response.choices[0].message + assert message is not None, f"reasoner choice has no message: {response}" + return message.reasoning_content + + +class TestDeepSeekReasoningDisable: + def test_reasoner_returns_reasoning_by_default( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + model = _register_reasoner(client, resources) + key = resources.key() + + response = unwrap( + client.gateway.chat( + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content=PROMPT)], + max_tokens=64, + ), + ) + ) + reasoning = _reasoning_content(response) + assert reasoning, ( + "control case: deepseek-reasoner returned no reasoning_content with no " + f"disable param, so the disable assertions below can't be trusted: {response}" + ) + + @pytest.mark.xfail( + strict=True, + reason=( + "LIT-3686 / GH #27453: DeepSeek reasoning_effort='none' and " + "thinking type='disabled' are silently dropped; reasoning not disabled" + ), + ) + def test_reasoning_effort_none_disables_reasoning( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + model = _register_reasoner(client, resources) + key = resources.key() + + response = unwrap( + client.gateway.chat( + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content=PROMPT)], + max_tokens=64, + reasoning_effort="none", + ), + ) + ) + assert not _reasoning_content(response), ( + "reasoning_effort='none' must disable reasoning, but reasoning_content " + f"is still present: {response}" + ) + + @pytest.mark.xfail( + strict=True, + reason=( + "LIT-3686 / GH #27453: DeepSeek reasoning_effort='none' and " + "thinking type='disabled' are silently dropped; reasoning not disabled" + ), + ) + def test_thinking_disabled_disables_reasoning( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + model = _register_reasoner(client, resources) + key = resources.key() + + response = unwrap( + client.gateway.chat( + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content=PROMPT)], + max_tokens=64, + thinking=ThinkingParam(type="disabled"), + ), + ) + ) + assert not _reasoning_content(response), ( + "thinking={'type': 'disabled'} must disable reasoning, but " + f"reasoning_content is still present: {response}" + ) diff --git a/tests/e2e/llm_translation/test_provider_features_e2e.py b/tests/e2e/llm_translation/test_provider_features_e2e.py new file mode 100644 index 00000000000..9c99c1be161 --- /dev/null +++ b/tests/e2e/llm_translation/test_provider_features_e2e.py @@ -0,0 +1,148 @@ +"""Live e2e for model-specific request features: service_tier and prompt caching. + +Each case asserts the feature took effect, not just a 200. + +service_tier is an OpenAI concept. The proxy forwards it and the provider echoes +the tier back on the response, so sending a non-default tier ("flex") and reading +it back off ``service_tier`` proves the param was honored end to end; litellm's own +default injection would report "default", so a "flex" echo can only come from the +request being forwarded. Bedrock and Vertex do not accept service_tier, so that +cell is OpenAI-only by design. + +Prompt caching is asserted through provider prompt-cache usage tokens. The +deterministic path is explicit ``cache_control`` on an Anthropic-family model +(here Bedrock's Claude): a large cacheable prefix is sent twice and the second +call must report ``cache_read_input_tokens > 0``. OpenAI and Gemini only offer +implicit automatic caching, which does not deterministically produce a cache read +within a test window (verified: repeated >3k-token prompts kept +``prompt_tokens_details.cached_tokens`` at 0), so those caching cells are out of +scope here and covered only by the explicit-cache-control Bedrock case. +""" + +from __future__ import annotations + +import pytest +from pydantic import BaseModel + +from e2e_config import unique_marker +from e2e_http import unwrap +from lifecycle import ResourceManager +from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody +from passthrough_client import PassthroughClient + +pytestmark = pytest.mark.e2e + +SERVICE_TIER = "flex" +CACHE_MIN_READ_TOKENS = 1 + + +class CacheControl(BaseModel): + type: str = "ephemeral" + + +class CacheTextBlock(BaseModel): + type: str = "text" + text: str + cache_control: CacheControl | None = None + + +class RichMessage(BaseModel): + role: str + content: list[CacheTextBlock] + + +class CacheChatBody(BaseModel): + model: str + messages: list[RichMessage] + max_tokens: int + + +def cacheable_prefix() -> str: + return ( + "You are a policy compliance auditor. The following corpus is the immutable " + "reference the assistant must consult on every turn. " + ) + ("Clause: obey all safety, formatting, and citation rules exactly. " * 400) + + +def post_chat(client: PassthroughClient, key: str, body: BaseModel) -> ChatResponse: + return unwrap( + client.gateway.transport.post( + "/chat/completions", + headers=client.gateway.transport.bearer(key), + json=body, + response_type=ChatResponse, + ) + ) + + +class TestServiceTier: + @pytest.mark.covers("llm.chat_completions.openai.service_tier.works", exercised_on=[]) + def test_openai_service_tier_is_echoed( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + model = f"e2e-service-tier-{unique_marker()}" + model_id = client.gateway.create_model( + model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY") + ) + resources.defer(lambda: client.gateway.delete_model(model_id)) + key = resources.key() + + response = unwrap( + client.gateway.chat( + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content="reply with one word")], + max_tokens=64, + service_tier=SERVICE_TIER, + ), + ) + ) + assert response.service_tier == SERVICE_TIER, ( + f"service_tier not honored: sent {SERVICE_TIER!r}, response reported " + f"{response.service_tier!r} ({response})" + ) + + +class TestPromptCaching: + @pytest.mark.covers( + "llm.chat_completions.bedrock_converse.prompt_cache_5m.nonstream.cache_hit", exercised_on=[] + ) + def test_bedrock_cache_control_produces_cache_read( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + model = f"e2e-bedrock-cache-{unique_marker()}" + model_id = client.gateway.create_model( + model, + LiteLLMParamsBody( + model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", + aws_region_name="us-east-1", + ), + ) + resources.defer(lambda: client.gateway.delete_model(model_id)) + key = resources.key() + + body = CacheChatBody( + model=model, + max_tokens=32, + messages=[ + RichMessage( + role="user", + content=[ + CacheTextBlock(text=cacheable_prefix(), cache_control=CacheControl()), + CacheTextBlock(text="Answer in one word: acknowledged?"), + ], + ) + ], + ) + + first = post_chat(client, key, body) + assert first.usage is not None, f"first call reported no usage: {first}" + + second = post_chat(client, key, body) + assert second.usage is not None, f"second call reported no usage: {second}" + cache_read = second.usage.cache_read_input_tokens + assert cache_read is not None and cache_read >= CACHE_MIN_READ_TOKENS, ( + "second identical request did not read the prompt cache: " + f"cache_read_input_tokens={cache_read!r} (usage={second.usage})" + ) diff --git a/tests/e2e/logging/conftest.py b/tests/e2e/logging/conftest.py new file mode 100644 index 00000000000..40c19aefca7 --- /dev/null +++ b/tests/e2e/logging/conftest.py @@ -0,0 +1,37 @@ +"""Fixtures for the Datadog logging suite. + +These tests drive the Datadog batch-send path (#25663) directly against the real +Datadog logs intake with synthetic events - no LLM calls, no proxy, no log +read-back - so they need only the shipping credentials DD_API_KEY + DD_SITE +(DD_SERVICE is an optional tag). No Datadog Application key is required, and they +skip when the shipping credentials are absent from the environment. +""" + +import os + +import pytest + +from logging_client import LoggingClient, build_logging_client + + +def pytest_configure(config: pytest.Config) -> None: + config.addinivalue_line( + "markers", + "covers: registry cell a test covers, e.g. logging.datadog.success.writes_object", + ) + + +@pytest.fixture(scope="session") +def client() -> LoggingClient: + """The logging suite's client: holds the shared Gateway so `resources` / + `scoped_key` clean up keys, and adds `/metrics` scraping.""" + return build_logging_client() + + +@pytest.fixture +def datadog_creds() -> None: + """Gate the suite on the Datadog shipping credentials. The DataDogLogger is built + inside each async test, not here, because its __init__ schedules a periodic-flush + task via asyncio.create_task and so needs a running event loop.""" + if not (os.getenv("DD_API_KEY") and os.getenv("DD_SITE")): + pytest.skip("set DD_API_KEY and DD_SITE to run the Datadog logging suite") diff --git a/tests/e2e/logging/logging_client.py b/tests/e2e/logging/logging_client.py new file mode 100644 index 00000000000..a3213fbdb00 --- /dev/null +++ b/tests/e2e/logging/logging_client.py @@ -0,0 +1,48 @@ +"""Client for the logging e2e suite: drive traffic and scrape the proxy's +Prometheus ``/metrics`` endpoint. + +Holds the shared Gateway so the ``resources`` fixture cleans up keys it creates. +``/metrics`` is exposed as plaintext (not a typed JSON body), so scraping goes +through ``transport.probe`` and returns the raw exposition text for a Prometheus +parser to read. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +from e2e_gateway import Gateway, build_gateway +from e2e_http import NoBody, unwrap +from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody + + +@dataclass(frozen=True, slots=True) +class LoggingClient: + gateway: Gateway + + def key_with_alias(self, alias: str, *, models: list[str]) -> str: + return self.gateway.generate_key( + KeyGenerateBody(key_alias=alias, models=models, user_id=f"e2e-{alias}") + ) + + def delete_key(self, key: str) -> None: + self.gateway.delete_key(key) + + def chat(self, key: str, model: str, text: str) -> ChatResponse: + return unwrap( + self.gateway.chat( + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content=text)], + max_tokens=64, + ), + ) + ) + + def scrape_metrics(self) -> str: + return self.gateway.probe("/metrics", params=NoBody()).body + + +def build_logging_client() -> LoggingClient: + return LoggingClient(gateway=build_gateway()) diff --git a/tests/e2e/logging/test_prometheus_cardinality_e2e.py b/tests/e2e/logging/test_prometheus_cardinality_e2e.py new file mode 100644 index 00000000000..163293a3009 --- /dev/null +++ b/tests/e2e/logging/test_prometheus_cardinality_e2e.py @@ -0,0 +1,70 @@ +"""Live e2e: Prometheus request metrics grow one series per virtual key. + +The proxy exposes ``/metrics`` (prometheus is in the callbacks and +``require_auth_for_metrics_endpoint`` is off in the e2e config). The counter +``litellm_requests_metric_total`` carries an ``api_key_alias`` label, so driving +traffic through keys with distinct aliases must produce a distinct labeled series +per alias. This is the per-key cardinality contract: a regression that stops +stamping ``api_key_alias`` (or collapses every key onto one series) would drop +the aliases and fail here. + +Scraping goes through ``transport.probe`` (raw text) and is parsed with +prometheus_client; the metric is eventually consistent (it increments on the +success-logging callback), so the scrape polls to a deadline. +""" + +from __future__ import annotations + +import time + +import pytest +from prometheus_client.parser import text_string_to_metric_families + +from e2e_config import unique_marker +from lifecycle import ResourceManager +from logging_client import LoggingClient + +pytestmark = pytest.mark.e2e + +DRIVER_MODEL = "gemini-2.5-flash" +REQUESTS_METRIC = "litellm_requests_metric_total" +ALIAS_LABEL = "api_key_alias" +DISTINCT_KEYS = 3 + + +def _aliases_in_metric(exposition: str, metric: str, label: str) -> frozenset[str]: + """The set of ``label`` values present on ``metric`` samples in a scrape.""" + return frozenset( + sample.labels[label] + for family in text_string_to_metric_families(exposition) + for sample in family.samples + if sample.name == metric and label in sample.labels + ) + + +class TestPrometheusPerKeyCardinality: + @pytest.mark.covers("logging.prometheus.success.exports_metric", exercised_on=[]) + def test_distinct_key_aliases_produce_distinct_series( + self, client: LoggingClient, resources: ResourceManager + ) -> None: + aliases = tuple(f"e2e-prom-{unique_marker()}" for _ in range(DISTINCT_KEYS)) + for alias in aliases: + key = client.key_with_alias(alias, models=[DRIVER_MODEL]) + resources.defer(lambda k=key: client.delete_key(k)) + response = client.chat(key, DRIVER_MODEL, f"reply with one word {alias}") + assert response.model, f"driver call for {alias} returned no model: {response}" + + wanted = frozenset(aliases) + deadline = time.monotonic() + client.gateway.poll_timeout + seen: frozenset[str] = frozenset() + while time.monotonic() < deadline: + seen = _aliases_in_metric(client.scrape_metrics(), REQUESTS_METRIC, ALIAS_LABEL) + if wanted <= seen: + break + time.sleep(client.gateway.poll_interval) + + missing = wanted - seen + assert not missing, ( + f"{REQUESTS_METRIC} is missing a per-key series for aliases {sorted(missing)}; " + f"each distinct {ALIAS_LABEL} must grow its own series" + ) diff --git a/tests/e2e/models.py b/tests/e2e/models.py index 9f013130a05..075880eb126 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -6,6 +6,8 @@ response validates without mirroring every proxy field. No untyped dicts. from __future__ import annotations +from typing import Literal + from pydantic import BaseModel, ConfigDict, RootModel # ---------- keys ---------- @@ -30,6 +32,7 @@ class KeyGenerateBody(BaseModel): user_id: str | None = None team_id: str | None = None budget_id: str | None = None + key_alias: str | None = None model_max_budget: dict[str, ModelBudgetEntry] | None = None budget_fallbacks: dict[str, list[str]] | None = None budget_limits: list[BudgetWindow] | None = None @@ -88,6 +91,16 @@ class ChatMessage(BaseModel): content: str +class ThinkingParam(BaseModel): + """Extended-thinking control shared by Anthropic and DeepSeek reasoner models. + DeepSeek accepts only ``type`` (enabled/disabled) and ignores budget_tokens; + Anthropic also honors budget_tokens. Sending ``type="disabled"`` is the + product-facing way a caller turns reasoning off (LIT-3686 / GH #27453).""" + + type: Literal["enabled", "disabled"] + budget_tokens: int | None = None + + class ChatBody(BaseModel): model: str messages: list[ChatMessage] @@ -95,6 +108,9 @@ class ChatBody(BaseModel): max_tokens: int | None = None user: str | None = None metadata: ChatMetadata | None = None + reasoning_effort: str | None = None + thinking: ThinkingParam | None = None + service_tier: str | None = None class AnthropicMessagesBody(BaseModel): @@ -105,16 +121,24 @@ class AnthropicMessagesBody(BaseModel): class OutMessage(BaseModel): content: str | None = None + reasoning_content: str | None = None class ChatChoice(BaseModel): message: OutMessage | None = None +class PromptTokensDetails(BaseModel): + cached_tokens: int | None = None + + class Usage(BaseModel): prompt_tokens: int | None = None completion_tokens: int | None = None total_tokens: int | None = None + cache_read_input_tokens: int | None = None + cache_creation_input_tokens: int | None = None + prompt_tokens_details: PromptTokensDetails | None = None class ChatResponse(BaseModel): @@ -122,6 +146,7 @@ class ChatResponse(BaseModel): model: str | None = None choices: list[ChatChoice] = [] usage: Usage | None = None + service_tier: str | None = None class EmbedBody(BaseModel): @@ -166,6 +191,7 @@ class OcrResponse(BaseModel): class SpendLogRow(BaseModel): request_id: str | None = None + api_key: str | None = None model: str | None = None spend: float | None = None status: str | None = None @@ -287,6 +313,32 @@ class ModelInfoResponse(BaseModel): data: list[ModelInfoEntry] = [] +class FileEntry(BaseModel): + id: str + + +class FileListResponse(BaseModel): + """GET /files answer. `data` is required on purpose: a 200 whose body lacks + the OpenAI-format file list must fail validation, not pass vacuously.""" + + data: list[FileEntry] + + +class FineTuningJobsParams(BaseModel): + custom_llm_provider: Literal["openai", "azure"] + + +class FineTuningJobEntry(BaseModel): + id: str + + +class FineTuningJobsResponse(BaseModel): + """GET /fine_tuning/jobs answer; `data` required for the same reason as + FileListResponse.""" + + data: list[FineTuningJobEntry] + + # ---------- model management ---------- @@ -301,12 +353,23 @@ class LiteLLMParamsBody(BaseModel): api_key: str | None = None api_base: str | None = None api_version: str | None = None + aws_region_name: str | None = None + vertex_project: str | None = None + vertex_location: str | None = None + vertex_credentials: str | None = None + bucket_name: str | None = None + s3_bucket_name: str | None = None + s3_region_name: str | None = None + s3_access_key_id: str | None = None + s3_secret_access_key: str | None = None + aws_batch_role_arn: str | None = None input_cost_per_token: float | None = None output_cost_per_token: float | None = None class ModelInfoBody(BaseModel): id: str + mode: Literal["batch", "realtime", "image_generation"] | None = None class ModelNewBody(BaseModel):