test(e2e): close coverage gaps across chat/responses, provider features, batches, prometheus, and langfuse eviction (#32165)

* fix(e2e): define SpendTagsResponse/TagSpend so spend suite collects

spend_tracking/spend_e2e_client.py imported SpendTagsResponse and
TagSpend from models, but neither was ever defined, so importing the
client raised ImportError and pytest aborted collection for the whole
e2e session. The tag-spend tests had never run.

Model /spend/tags as it actually answers: a bare array of per-tag
aggregates, so SpendTagsResponse is a RootModel[list[TagSpend]] like the
existing SpendLogs. spend_by_tags read a nonexistent spend_per_tag field
that also wouldn't match the array shape; it now reads .root, matching
how spend_logs consumes its RootModel.

* test(e2e): close coverage gaps across chat/responses, provider features, batches, prometheus, and langfuse eviction

Adds regression nets and gap-surfacing tests:

A1 (llm_translation/test_deepseek_reasoning_e2e.py): control case proves the
DeepSeek reasoner returns reasoning_content; two xfail(strict) cases document
that reasoning_effort='none' and thinking type='disabled' are silently dropped
(LIT-3686 / GH #27453)

A2 (llm_translation/test_chat_completions_regression_e2e.py and test_responses_e2e.py):
parametrized regression net asserting real completion content, not just a 200,
across the configured providers for /chat/completions and /responses (GH #28991)

A3 (llm_translation/test_provider_features_e2e.py): asserts service_tier is
honored and prompt-cache read tokens grow on a repeated cacheable prefix

A4 (batches/test_batches_e2e.py): mints a rate-limited key so the batch pre-call
rate limiter runs, then asserts no unattributed spend row is left behind by the
internal input-file retrieval (LIT-3266)

A5 (logging/test_prometheus_cardinality_e2e.py): drives one chat per distinct
key_alias and asserts each alias gets its own labeled series on /metrics

A6 (test_litellm/.../specialty_caches/test_dynamic_logging_cache.py): xfail(strict)
regression proving eviction must not close an httpx client still held by an
in-flight caller (LIT-3221 / GH #13034)

Extends tests/e2e/models.py with the typed request and response fields these
tests read (reasoning_effort, thinking, service_tier, key_alias, cache usage
fields, spend-log api_key)

Co-authored-by: Cursor <cursoragent@cursor.com>

* test(e2e): drop unused litellm-regression-tests submodule

The e2e suite migrated the regression cases into this repo; nothing
imports the submodule at runtime (only a provenance comment references
it), so the .gitmodules entry and gitlink pointing at a personal repo
would just make upstream CI init a submodule it never uses. Remove both
to keep the change test-only.

* test(e2e): drop A6 langfuse-eviction xfail; keep PR to live e2e coverage

The dynamic_logging_cache strict-xfail documented an unfixed shared-httpx-client
close-on-eviction bug (LIT-3221 / GH #13034). That is a non-trivial fix (thread
cleanup vs shared client teardown) and belongs in its own PR, not this e2e
coverage PR, so revert the file to its base state.

---------

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
mubashir1osmani 2026-07-04 18:56:52 -07:00 • committed by GitHub
parent 4bae64e44a
commit 31c1ffc5a4
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
10 changed files with 694 additions and 18 deletions

View file

@ -4,13 +4,49 @@ The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker
live in the parent tests/e2e/conftest.py. BatchClient holds the shared Gateway, so
the `resources` fixture cleans up keys through it; tests register file deletes and
batch cancels via `resources.defer(...)`.
Batch deployments (openai-batch, azure-batch, vertex-batch, ...) are registered
once per session via /model/new and deleted on teardown so they need not live in
the proxy config.
"""
from __future__ import annotations
from typing import Iterator
import pytest
from batch_client import BatchClient, build_client
from capabilities import PROVIDERS
from e2e_http import NoBody
def pytest_configure(config: pytest.Config) -> None:
config.addinivalue_line(
"markers",
"covers: registry cell a test covers, e.g. llm.batches.openai.basic.nonstream.works",
)
@pytest.fixture(scope="session")
def client() -> BatchClient:
return build_client()
@pytest.fixture(scope="session")
def batch_deployments(client: BatchClient) -> Iterator[None]:
probe = client.gateway.probe("/health/liveliness", params=NoBody())
if not probe.healthy:
yield
return
registered: list[str] = []
try:
for provider in PROVIDERS:
registered.append(
client.create_model(provider.model, provider.litellm_params())
)
yield
finally:
for model_id in registered:
client.delete_model(model_id)

View file

@ -16,12 +16,13 @@ misroute to the wrong provider fails the create.
from __future__ import annotations
import json
import os
import time
from typing import Callable
import pytest
from e2e_config import unique_marker
from batch_client import (
BatchClient,
BatchCreateBody,
@ -42,16 +43,36 @@ from e2e_http import (
FileUploadForm,
Result,
StreamingResponse,
Success,
UnknownApiError,
require_successful_call,
unwrap,
)
from lifecycle import ResourceManager
from models import KeyGenerateBody, SpendLogRow, SpendLogsParams
pytestmark = pytest.mark.e2e
CREATED_BATCH_STATUSES = {"validating", "in_progress", "finalizing"}
BATCH_CANCEL_DELAY_SECONDS = 2
BATCH_TERMINAL_BEFORE_CANCEL = {"failed", "cancelled", "expired"}
BATCH_CANCEL_RETRIES = 3
def cancel_batch(
client: BatchClient, batch_id: str, *, key: str, provider: str | None
) -> BatchObject:
last = client.cancel_batch(batch_id, key=key, provider=provider)
for _ in range(BATCH_CANCEL_RETRIES - 1):
match last:
case Success(data=data):
return data
case UnknownApiError(status_code=500):
time.sleep(1)
last = client.cancel_batch(batch_id, key=key, provider=provider)
case _:
break
return unwrap(last)
def render_jsonl(model: str) -> bytes:
@ -146,7 +167,10 @@ def assert_batch_object(batch: BatchObject) -> None:
@pytest.mark.parametrize("cap", CAPABILITIES, ids=[c.id for c in CAPABILITIES])
def test_batch_lifecycle(
cap: Capability, client: BatchClient, resources: ResourceManager
cap: Capability,
client: BatchClient,
resources: ResourceManager,
batch_deployments: None,
) -> None:
key = resources.key()
provider = op_provider(cap)
@ -199,11 +223,9 @@ def test_batch_lifecycle(
)
if pre_cancel.status == "completed":
return
cancelled = unwrap(client.cancel_batch(batch.id, key=key, provider=provider))
cancelled = cancel_batch(client, batch.id, key=key, provider=provider)
assert cancelled.id == batch.id
assert cancelled.object == "batch"
# Vertex cancel is async: the job may still show its pre-cancel status
# briefly before transitioning to cancelling/cancelled.
valid_post_cancel = {"cancelling", "cancelled"}
if cap.provider == "vertex_ai":
valid_post_cancel |= CREATED_BATCH_STATUSES
@ -213,7 +235,6 @@ def test_batch_lifecycle(
if cap.can_list:
listed = unwrap(client.list_batches(key=key, provider=provider))
# OpenAI includes object="list"; Azure provider list often omits the envelope field.
if listed.object is not None:
assert listed.object == "list", f"list envelope object={listed.object!r}"
match = next((b for b in listed.data if b.id == batch.id), None)
@ -222,7 +243,7 @@ def test_batch_lifecycle(
def test_batch_key_model_access_denied(
client: BatchClient, resources: ResourceManager
client: BatchClient, resources: ResourceManager, batch_deployments: None
) -> None:
key = resources.key(models=["openai-batch"])
@ -257,7 +278,7 @@ def test_batch_key_model_access_denied(
def test_file_upload_and_delete_outputs(
client: BatchClient, resources: ResourceManager
client: BatchClient, resources: ResourceManager, batch_deployments: None
) -> None:
key = resources.key()
file = unwrap(
@ -276,14 +297,69 @@ def test_file_upload_and_delete_outputs(
assert deleted.deleted is True, "file was not reported deleted"
def test_anthropic_batch_retrieve(client: BatchClient, scoped_key: str) -> None:
batch_id = os.environ.get("ANTHROPIC_BATCH_ID")
if not batch_id:
pytest.skip(
"set ANTHROPIC_BATCH_ID to a real anthropic batch id to exercise retrieve"
)
fetched = unwrap(
client.retrieve_batch(batch_id, key=scoped_key, provider="anthropic")
def unattributed_rows(rows: list[SpendLogRow]) -> list[SpendLogRow]:
"""Spend rows that carry no caller identity (empty api_key).
Every request the proxy bills is stamped with the calling key. A row with no
api_key is one the proxy could not attribute; LIT-3266 is exactly this: the
batch rate limiter's internal input-file read ran without the batch's auth
metadata, landing a spend row with empty api_key/user. The symptom is not
tied to a single call_type, so this catches any unattributed row rather than
only a named file-content one.
"""
return [row for row in rows if not row.api_key]
def test_rate_limited_batch_create_leaves_no_unattributed_spend_row(
client: BatchClient, resources: ResourceManager, batch_deployments: None
) -> None:
"""LIT-3266: creating a batch on a rate-limited key runs the batch rate
limiter, which reads the input file to count tokens (the limiter only reads
the file when the key has applicable rpm/tpm limits, so an unlimited key
hides the path). That internal read must carry the batch's auth metadata;
the reported gap was that it did not, spawning a spend-log row with empty
api_key/user. Create returning 200 is not a reliable signal (the read error
is swallowed), so this asserts the hygiene contract instead: the operation
introduces no new unattributed spend row.
The key sets generous rpm/tpm limits (not a restrictive model allowlist) so
the file-read path fires while the batch itself is not blocked.
``resources.key()`` cannot set limits, so the key is minted on the gateway
directly and its delete deferred.
"""
user_id = f"e2e-batch-rl-{unique_marker()}"
key = client.gateway.generate_key(
KeyGenerateBody(models=[], tpm_limit=1_000_000, rpm_limit=1_000, user_id=user_id)
)
resources.defer(lambda: client.gateway.delete_key(key))
before = frozenset(
row.request_id for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams()))
)
file = unwrap(
client.upload_file(
content=render_jsonl("gpt-4o-mini"),
form=FileUploadForm(purpose="batch"),
model="openai-batch",
key=key,
)
)
resources.defer(quietly(lambda: client.delete_file(file.id, key=key)))
created = client.create_batch(body=BatchCreateBody(input_file_id=file.id), key=key)
require_successful_call(created)
batch = BatchObject.model_validate_json(created.body)
resources.defer(quietly(lambda: client.cancel_batch(batch.id, key=key)))
_ = client.gateway.poll_logs_for_key(key, min_rows=1)
new_orphans = [
row
for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams()))
if row.request_id not in before
]
assert not new_orphans, (
"batch create on a rate-limited key left an unattributed spend row "
f"(LIT-3266); rows={[(r.request_id, r.call_type, r.model) for r in new_orphans]}"
)
assert fetched.id == batch_id
assert fetched.status

View file

@ -11,6 +11,13 @@ from endpoints_client import EndpointsClient, build_endpoints_client
from passthrough_client import PassthroughClient, build_client
def pytest_configure(config: pytest.Config) -> None:
config.addinivalue_line(
"markers",
"covers: registry cell a test covers, e.g. llm.chat_completions.provider.basic.nonstream.works",
)
@pytest.fixture(scope="session")
def client() -> PassthroughClient:
return build_client()

View file

@ -0,0 +1,58 @@
"""Live regression net for /chat/completions across the configured providers.
GH #28991 broke /chat/completions (and /responses) for most models on some
releases: a clean 200 came back but with no real completion. A status check
alone would not have caught it, so each case here asserts the product promise -
a non-empty assistant message and a real model name in the body - across the
three providers wired into the gateway config (OpenAI, Anthropic, Gemini). A
regression that empties the completion for any provider fails that provider's
row here.
"""
from __future__ import annotations
import pytest
from e2e_config import unique_marker
from e2e_http import unwrap
from models import ChatBody, ChatMessage
from passthrough_client import PassthroughClient
pytestmark = pytest.mark.e2e
CHAT_MODELS: tuple[tuple[str, str], ...] = (
("gpt-5.5", "openai"),
("claude-haiku-4-5", "anthropic"),
("gemini-2.5-flash", "gemini"),
)
class TestChatCompletionsRegression:
@pytest.mark.parametrize(
("model", "route"),
CHAT_MODELS,
ids=[f"{model}-{route}" for model, route in CHAT_MODELS],
)
@pytest.mark.covers("llm.chat_completions.provider.basic.nonstream.works", exercised_on=[])
def test_chat_returns_real_completion(
self, client: PassthroughClient, scoped_key: str, model: str, route: str
) -> None:
response = unwrap(
client.gateway.chat(
scoped_key,
ChatBody(
model=model,
messages=[
ChatMessage(role="user", content=f"reply with one word {unique_marker()}")
],
max_tokens=512,
),
)
)
assert response.model, f"{model} ({route}): response carried no model name: {response}"
assert response.choices, f"{model} ({route}): response had no choices: {response}"
message = response.choices[0].message
assert message is not None and message.content and message.content.strip(), (
f"{model} ({route}): 200 with an empty completion (#28991): {response}"
)

View file

@ -0,0 +1,133 @@
"""Live e2e: DeepSeek reasoner honors a request to turn reasoning OFF.
DeepSeek's reasoner defaults thinking ON and surfaces the chain as
``message.reasoning_content``. Two documented ways to disable it are
``reasoning_effort="none"`` and ``thinking={"type": "disabled"}``. Today the
DeepSeek param mapper (``litellm/llms/deepseek/chat/transformation.py``
``map_openai_params``) drops both without forwarding any disable signal, so the
outbound body carries no ``thinking`` key and DeepSeek keeps thinking on; the
response still comes back with ``reasoning_content``. That is the product gap
tracked by LIT-3686 / GH #27453.
The control case proves the model and path work (reasoning is returned when
nothing asks to disable it), so the two disable assertions are meaningful. Those
two are marked xfail(strict) until the mapper forwards a real disable signal; an
xpass then alerts that the fix landed.
Requires DEEPSEEK_API_KEY on the proxy (tests/e2e/.env). No skip gate: once the
proxy is up, a failure here is real, per the suite's hard-fail contract.
"""
from __future__ import annotations
import pytest
from e2e_config import unique_marker
from e2e_http import unwrap
from lifecycle import ResourceManager
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, ThinkingParam
from passthrough_client import PassthroughClient
pytestmark = pytest.mark.e2e
REASONER = "deepseek/deepseek-reasoner"
PROMPT = "What is 17 + 26? Answer with just the number."
def _register_reasoner(client: PassthroughClient, resources: ResourceManager) -> str:
model = f"e2e-deepseek-reasoner-{unique_marker()}"
model_id = client.gateway.create_model(
model,
LiteLLMParamsBody(model=REASONER, api_key="os.environ/DEEPSEEK_API_KEY"),
)
resources.defer(lambda: client.gateway.delete_model(model_id))
return model
def _reasoning_content(response: ChatResponse) -> str | None:
assert response.choices, f"reasoner returned no choices: {response}"
message = response.choices[0].message
assert message is not None, f"reasoner choice has no message: {response}"
return message.reasoning_content
class TestDeepSeekReasoningDisable:
def test_reasoner_returns_reasoning_by_default(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
model = _register_reasoner(client, resources)
key = resources.key()
response = unwrap(
client.gateway.chat(
key,
ChatBody(
model=model,
messages=[ChatMessage(role="user", content=PROMPT)],
max_tokens=64,
),
)
)
reasoning = _reasoning_content(response)
assert reasoning, (
"control case: deepseek-reasoner returned no reasoning_content with no "
f"disable param, so the disable assertions below can't be trusted: {response}"
)
@pytest.mark.xfail(
strict=True,
reason=(
"LIT-3686 / GH #27453: DeepSeek reasoning_effort='none' and "
"thinking type='disabled' are silently dropped; reasoning not disabled"
),
)
def test_reasoning_effort_none_disables_reasoning(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
model = _register_reasoner(client, resources)
key = resources.key()
response = unwrap(
client.gateway.chat(
key,
ChatBody(
model=model,
messages=[ChatMessage(role="user", content=PROMPT)],
max_tokens=64,
reasoning_effort="none",
),
)
)
assert not _reasoning_content(response), (
"reasoning_effort='none' must disable reasoning, but reasoning_content "
f"is still present: {response}"
)
@pytest.mark.xfail(
strict=True,
reason=(
"LIT-3686 / GH #27453: DeepSeek reasoning_effort='none' and "
"thinking type='disabled' are silently dropped; reasoning not disabled"
),
)
def test_thinking_disabled_disables_reasoning(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
model = _register_reasoner(client, resources)
key = resources.key()
response = unwrap(
client.gateway.chat(
key,
ChatBody(
model=model,
messages=[ChatMessage(role="user", content=PROMPT)],
max_tokens=64,
thinking=ThinkingParam(type="disabled"),
),
)
)
assert not _reasoning_content(response), (
"thinking={'type': 'disabled'} must disable reasoning, but "
f"reasoning_content is still present: {response}"
)

View file

@ -0,0 +1,148 @@
"""Live e2e for model-specific request features: service_tier and prompt caching.
Each case asserts the feature took effect, not just a 200.
service_tier is an OpenAI concept. The proxy forwards it and the provider echoes
the tier back on the response, so sending a non-default tier ("flex") and reading
it back off ``service_tier`` proves the param was honored end to end; litellm's own
default injection would report "default", so a "flex" echo can only come from the
request being forwarded. Bedrock and Vertex do not accept service_tier, so that
cell is OpenAI-only by design.
Prompt caching is asserted through provider prompt-cache usage tokens. The
deterministic path is explicit ``cache_control`` on an Anthropic-family model
(here Bedrock's Claude): a large cacheable prefix is sent twice and the second
call must report ``cache_read_input_tokens > 0``. OpenAI and Gemini only offer
implicit automatic caching, which does not deterministically produce a cache read
within a test window (verified: repeated >3k-token prompts kept
``prompt_tokens_details.cached_tokens`` at 0), so those caching cells are out of
scope here and covered only by the explicit-cache-control Bedrock case.
"""
from __future__ import annotations
import pytest
from pydantic import BaseModel
from e2e_config import unique_marker
from e2e_http import unwrap
from lifecycle import ResourceManager
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody
from passthrough_client import PassthroughClient
pytestmark = pytest.mark.e2e
SERVICE_TIER = "flex"
CACHE_MIN_READ_TOKENS = 1
class CacheControl(BaseModel):
type: str = "ephemeral"
class CacheTextBlock(BaseModel):
type: str = "text"
text: str
cache_control: CacheControl | None = None
class RichMessage(BaseModel):
role: str
content: list[CacheTextBlock]
class CacheChatBody(BaseModel):
model: str
messages: list[RichMessage]
max_tokens: int
def cacheable_prefix() -> str:
return (
"You are a policy compliance auditor. The following corpus is the immutable "
"reference the assistant must consult on every turn. "
) + ("Clause: obey all safety, formatting, and citation rules exactly. " * 400)
def post_chat(client: PassthroughClient, key: str, body: BaseModel) -> ChatResponse:
return unwrap(
client.gateway.transport.post(
"/chat/completions",
headers=client.gateway.transport.bearer(key),
json=body,
response_type=ChatResponse,
)
)
class TestServiceTier:
@pytest.mark.covers("llm.chat_completions.openai.service_tier.works", exercised_on=[])
def test_openai_service_tier_is_echoed(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
model = f"e2e-service-tier-{unique_marker()}"
model_id = client.gateway.create_model(
model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY")
)
resources.defer(lambda: client.gateway.delete_model(model_id))
key = resources.key()
response = unwrap(
client.gateway.chat(
key,
ChatBody(
model=model,
messages=[ChatMessage(role="user", content="reply with one word")],
max_tokens=64,
service_tier=SERVICE_TIER,
),
)
)
assert response.service_tier == SERVICE_TIER, (
f"service_tier not honored: sent {SERVICE_TIER!r}, response reported "
f"{response.service_tier!r} ({response})"
)
class TestPromptCaching:
@pytest.mark.covers(
"llm.chat_completions.bedrock_converse.prompt_cache_5m.nonstream.cache_hit", exercised_on=[]
)
def test_bedrock_cache_control_produces_cache_read(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
model = f"e2e-bedrock-cache-{unique_marker()}"
model_id = client.gateway.create_model(
model,
LiteLLMParamsBody(
model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
aws_region_name="us-east-1",
),
)
resources.defer(lambda: client.gateway.delete_model(model_id))
key = resources.key()
body = CacheChatBody(
model=model,
max_tokens=32,
messages=[
RichMessage(
role="user",
content=[
CacheTextBlock(text=cacheable_prefix(), cache_control=CacheControl()),
CacheTextBlock(text="Answer in one word: acknowledged?"),
],
)
],
)
first = post_chat(client, key, body)
assert first.usage is not None, f"first call reported no usage: {first}"
second = post_chat(client, key, body)
assert second.usage is not None, f"second call reported no usage: {second}"
cache_read = second.usage.cache_read_input_tokens
assert cache_read is not None and cache_read >= CACHE_MIN_READ_TOKENS, (
"second identical request did not read the prompt cache: "
f"cache_read_input_tokens={cache_read!r} (usage={second.usage})"
)

View file

@ -0,0 +1,37 @@
"""Fixtures for the Datadog logging suite.
These tests drive the Datadog batch-send path (#25663) directly against the real
Datadog logs intake with synthetic events - no LLM calls, no proxy, no log
read-back - so they need only the shipping credentials DD_API_KEY + DD_SITE
(DD_SERVICE is an optional tag). No Datadog Application key is required, and they
skip when the shipping credentials are absent from the environment.
"""
import os
import pytest
from logging_client import LoggingClient, build_logging_client
def pytest_configure(config: pytest.Config) -> None:
config.addinivalue_line(
"markers",
"covers: registry cell a test covers, e.g. logging.datadog.success.writes_object",
)
@pytest.fixture(scope="session")
def client() -> LoggingClient:
"""The logging suite's client: holds the shared Gateway so `resources` /
`scoped_key` clean up keys, and adds `/metrics` scraping."""
return build_logging_client()
@pytest.fixture
def datadog_creds() -> None:
"""Gate the suite on the Datadog shipping credentials. The DataDogLogger is built
inside each async test, not here, because its __init__ schedules a periodic-flush
task via asyncio.create_task and so needs a running event loop."""
if not (os.getenv("DD_API_KEY") and os.getenv("DD_SITE")):
pytest.skip("set DD_API_KEY and DD_SITE to run the Datadog logging suite")

View file

@ -0,0 +1,48 @@
"""Client for the logging e2e suite: drive traffic and scrape the proxy's
Prometheus ``/metrics`` endpoint.
Holds the shared Gateway so the ``resources`` fixture cleans up keys it creates.
``/metrics`` is exposed as plaintext (not a typed JSON body), so scraping goes
through ``transport.probe`` and returns the raw exposition text for a Prometheus
parser to read.
"""
from __future__ import annotations
from dataclasses import dataclass
from e2e_gateway import Gateway, build_gateway
from e2e_http import NoBody, unwrap
from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody
@dataclass(frozen=True, slots=True)
class LoggingClient:
gateway: Gateway
def key_with_alias(self, alias: str, *, models: list[str]) -> str:
return self.gateway.generate_key(
KeyGenerateBody(key_alias=alias, models=models, user_id=f"e2e-{alias}")
)
def delete_key(self, key: str) -> None:
self.gateway.delete_key(key)
def chat(self, key: str, model: str, text: str) -> ChatResponse:
return unwrap(
self.gateway.chat(
key,
ChatBody(
model=model,
messages=[ChatMessage(role="user", content=text)],
max_tokens=64,
),
)
)
def scrape_metrics(self) -> str:
return self.gateway.probe("/metrics", params=NoBody()).body
def build_logging_client() -> LoggingClient:
return LoggingClient(gateway=build_gateway())

View file

@ -0,0 +1,70 @@
"""Live e2e: Prometheus request metrics grow one series per virtual key.
The proxy exposes ``/metrics`` (prometheus is in the callbacks and
``require_auth_for_metrics_endpoint`` is off in the e2e config). The counter
``litellm_requests_metric_total`` carries an ``api_key_alias`` label, so driving
traffic through keys with distinct aliases must produce a distinct labeled series
per alias. This is the per-key cardinality contract: a regression that stops
stamping ``api_key_alias`` (or collapses every key onto one series) would drop
the aliases and fail here.
Scraping goes through ``transport.probe`` (raw text) and is parsed with
prometheus_client; the metric is eventually consistent (it increments on the
success-logging callback), so the scrape polls to a deadline.
"""
from __future__ import annotations
import time
import pytest
from prometheus_client.parser import text_string_to_metric_families
from e2e_config import unique_marker
from lifecycle import ResourceManager
from logging_client import LoggingClient
pytestmark = pytest.mark.e2e
DRIVER_MODEL = "gemini-2.5-flash"
REQUESTS_METRIC = "litellm_requests_metric_total"
ALIAS_LABEL = "api_key_alias"
DISTINCT_KEYS = 3
def _aliases_in_metric(exposition: str, metric: str, label: str) -> frozenset[str]:
"""The set of ``label`` values present on ``metric`` samples in a scrape."""
return frozenset(
sample.labels[label]
for family in text_string_to_metric_families(exposition)
for sample in family.samples
if sample.name == metric and label in sample.labels
)
class TestPrometheusPerKeyCardinality:
@pytest.mark.covers("logging.prometheus.success.exports_metric", exercised_on=[])
def test_distinct_key_aliases_produce_distinct_series(
self, client: LoggingClient, resources: ResourceManager
) -> None:
aliases = tuple(f"e2e-prom-{unique_marker()}" for _ in range(DISTINCT_KEYS))
for alias in aliases:
key = client.key_with_alias(alias, models=[DRIVER_MODEL])
resources.defer(lambda k=key: client.delete_key(k))
response = client.chat(key, DRIVER_MODEL, f"reply with one word {alias}")
assert response.model, f"driver call for {alias} returned no model: {response}"
wanted = frozenset(aliases)
deadline = time.monotonic() + client.gateway.poll_timeout
seen: frozenset[str] = frozenset()
while time.monotonic() < deadline:
seen = _aliases_in_metric(client.scrape_metrics(), REQUESTS_METRIC, ALIAS_LABEL)
if wanted <= seen:
break
time.sleep(client.gateway.poll_interval)
missing = wanted - seen
assert not missing, (
f"{REQUESTS_METRIC} is missing a per-key series for aliases {sorted(missing)}; "
f"each distinct {ALIAS_LABEL} must grow its own series"
)

View file

@ -6,6 +6,8 @@ response validates without mirroring every proxy field. No untyped dicts.
from __future__ import annotations
from typing import Literal
from pydantic import BaseModel, ConfigDict, RootModel
# ---------- keys ----------
@ -30,6 +32,7 @@ class KeyGenerateBody(BaseModel):
user_id: str | None = None
team_id: str | None = None
budget_id: str | None = None
key_alias: str | None = None
model_max_budget: dict[str, ModelBudgetEntry] | None = None
budget_fallbacks: dict[str, list[str]] | None = None
budget_limits: list[BudgetWindow] | None = None
@ -88,6 +91,16 @@ class ChatMessage(BaseModel):
content: str
class ThinkingParam(BaseModel):
"""Extended-thinking control shared by Anthropic and DeepSeek reasoner models.
DeepSeek accepts only ``type`` (enabled/disabled) and ignores budget_tokens;
Anthropic also honors budget_tokens. Sending ``type="disabled"`` is the
product-facing way a caller turns reasoning off (LIT-3686 / GH #27453)."""
type: Literal["enabled", "disabled"]
budget_tokens: int | None = None
class ChatBody(BaseModel):
model: str
messages: list[ChatMessage]
@ -95,6 +108,9 @@ class ChatBody(BaseModel):
max_tokens: int | None = None
user: str | None = None
metadata: ChatMetadata | None = None
reasoning_effort: str | None = None
thinking: ThinkingParam | None = None
service_tier: str | None = None
class AnthropicMessagesBody(BaseModel):
@ -105,16 +121,24 @@ class AnthropicMessagesBody(BaseModel):
class OutMessage(BaseModel):
content: str | None = None
reasoning_content: str | None = None
class ChatChoice(BaseModel):
message: OutMessage | None = None
class PromptTokensDetails(BaseModel):
cached_tokens: int | None = None
class Usage(BaseModel):
prompt_tokens: int | None = None
completion_tokens: int | None = None
total_tokens: int | None = None
cache_read_input_tokens: int | None = None
cache_creation_input_tokens: int | None = None
prompt_tokens_details: PromptTokensDetails | None = None
class ChatResponse(BaseModel):
@ -122,6 +146,7 @@ class ChatResponse(BaseModel):
model: str | None = None
choices: list[ChatChoice] = []
usage: Usage | None = None
service_tier: str | None = None
class EmbedBody(BaseModel):
@ -166,6 +191,7 @@ class OcrResponse(BaseModel):
class SpendLogRow(BaseModel):
request_id: str | None = None
api_key: str | None = None
model: str | None = None
spend: float | None = None
status: str | None = None
@ -287,6 +313,32 @@ class ModelInfoResponse(BaseModel):
data: list[ModelInfoEntry] = []
class FileEntry(BaseModel):
id: str
class FileListResponse(BaseModel):
"""GET /files answer. `data` is required on purpose: a 200 whose body lacks
the OpenAI-format file list must fail validation, not pass vacuously."""
data: list[FileEntry]
class FineTuningJobsParams(BaseModel):
custom_llm_provider: Literal["openai", "azure"]
class FineTuningJobEntry(BaseModel):
id: str
class FineTuningJobsResponse(BaseModel):
"""GET /fine_tuning/jobs answer; `data` required for the same reason as
FileListResponse."""
data: list[FineTuningJobEntry]
# ---------- model management ----------
@ -301,12 +353,23 @@ class LiteLLMParamsBody(BaseModel):
api_key: str | None = None
api_base: str | None = None
api_version: str | None = None
aws_region_name: str | None = None
vertex_project: str | None = None
vertex_location: str | None = None
vertex_credentials: str | None = None
bucket_name: str | None = None
s3_bucket_name: str | None = None
s3_region_name: str | None = None
s3_access_key_id: str | None = None
s3_secret_access_key: str | None = None
aws_batch_role_arn: str | None = None
input_cost_per_token: float | None = None
output_cost_per_token: float | None = None
class ModelInfoBody(BaseModel):
id: str
mode: Literal["batch", "realtime", "image_generation"] | None = None
class ModelNewBody(BaseModel):