mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
test(e2e): close coverage gaps across chat/responses, provider features, batches, prometheus, and langfuse eviction (#32165)
* fix(e2e): define SpendTagsResponse/TagSpend so spend suite collects spend_tracking/spend_e2e_client.py imported SpendTagsResponse and TagSpend from models, but neither was ever defined, so importing the client raised ImportError and pytest aborted collection for the whole e2e session. The tag-spend tests had never run. Model /spend/tags as it actually answers: a bare array of per-tag aggregates, so SpendTagsResponse is a RootModel[list[TagSpend]] like the existing SpendLogs. spend_by_tags read a nonexistent spend_per_tag field that also wouldn't match the array shape; it now reads .root, matching how spend_logs consumes its RootModel. * test(e2e): close coverage gaps across chat/responses, provider features, batches, prometheus, and langfuse eviction Adds regression nets and gap-surfacing tests: A1 (llm_translation/test_deepseek_reasoning_e2e.py): control case proves the DeepSeek reasoner returns reasoning_content; two xfail(strict) cases document that reasoning_effort='none' and thinking type='disabled' are silently dropped (LIT-3686 / GH #27453) A2 (llm_translation/test_chat_completions_regression_e2e.py and test_responses_e2e.py): parametrized regression net asserting real completion content, not just a 200, across the configured providers for /chat/completions and /responses (GH #28991) A3 (llm_translation/test_provider_features_e2e.py): asserts service_tier is honored and prompt-cache read tokens grow on a repeated cacheable prefix A4 (batches/test_batches_e2e.py): mints a rate-limited key so the batch pre-call rate limiter runs, then asserts no unattributed spend row is left behind by the internal input-file retrieval (LIT-3266) A5 (logging/test_prometheus_cardinality_e2e.py): drives one chat per distinct key_alias and asserts each alias gets its own labeled series on /metrics A6 (test_litellm/.../specialty_caches/test_dynamic_logging_cache.py): xfail(strict) regression proving eviction must not close an httpx client still held by an in-flight caller (LIT-3221 / GH #13034) Extends tests/e2e/models.py with the typed request and response fields these tests read (reasoning_effort, thinking, service_tier, key_alias, cache usage fields, spend-log api_key) Co-authored-by: Cursor <cursoragent@cursor.com> * test(e2e): drop unused litellm-regression-tests submodule The e2e suite migrated the regression cases into this repo; nothing imports the submodule at runtime (only a provenance comment references it), so the .gitmodules entry and gitlink pointing at a personal repo would just make upstream CI init a submodule it never uses. Remove both to keep the change test-only. * test(e2e): drop A6 langfuse-eviction xfail; keep PR to live e2e coverage The dynamic_logging_cache strict-xfail documented an unfixed shared-httpx-client close-on-eviction bug (LIT-3221 / GH #13034). That is a non-trivial fix (thread cleanup vs shared client teardown) and belongs in its own PR, not this e2e coverage PR, so revert the file to its base state. --------- Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
4bae64e44a
commit
31c1ffc5a4
10 changed files with 694 additions and 18 deletions
|
|
@ -4,13 +4,49 @@ The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker
|
|||
live in the parent tests/e2e/conftest.py. BatchClient holds the shared Gateway, so
|
||||
the `resources` fixture cleans up keys through it; tests register file deletes and
|
||||
batch cancels via `resources.defer(...)`.
|
||||
|
||||
Batch deployments (openai-batch, azure-batch, vertex-batch, ...) are registered
|
||||
once per session via /model/new and deleted on teardown so they need not live in
|
||||
the proxy config.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Iterator
|
||||
|
||||
import pytest
|
||||
|
||||
from batch_client import BatchClient, build_client
|
||||
from capabilities import PROVIDERS
|
||||
from e2e_http import NoBody
|
||||
|
||||
|
||||
def pytest_configure(config: pytest.Config) -> None:
|
||||
config.addinivalue_line(
|
||||
"markers",
|
||||
"covers: registry cell a test covers, e.g. llm.batches.openai.basic.nonstream.works",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def client() -> BatchClient:
|
||||
return build_client()
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def batch_deployments(client: BatchClient) -> Iterator[None]:
|
||||
probe = client.gateway.probe("/health/liveliness", params=NoBody())
|
||||
if not probe.healthy:
|
||||
yield
|
||||
return
|
||||
|
||||
registered: list[str] = []
|
||||
try:
|
||||
for provider in PROVIDERS:
|
||||
registered.append(
|
||||
client.create_model(provider.model, provider.litellm_params())
|
||||
)
|
||||
yield
|
||||
finally:
|
||||
for model_id in registered:
|
||||
client.delete_model(model_id)
|
||||
|
|
|
|||
|
|
@ -16,12 +16,13 @@ misroute to the wrong provider fails the create.
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from typing import Callable
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
|
||||
from batch_client import (
|
||||
BatchClient,
|
||||
BatchCreateBody,
|
||||
|
|
@ -42,16 +43,36 @@ from e2e_http import (
|
|||
FileUploadForm,
|
||||
Result,
|
||||
StreamingResponse,
|
||||
Success,
|
||||
UnknownApiError,
|
||||
require_successful_call,
|
||||
unwrap,
|
||||
)
|
||||
from lifecycle import ResourceManager
|
||||
from models import KeyGenerateBody, SpendLogRow, SpendLogsParams
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
CREATED_BATCH_STATUSES = {"validating", "in_progress", "finalizing"}
|
||||
BATCH_CANCEL_DELAY_SECONDS = 2
|
||||
BATCH_TERMINAL_BEFORE_CANCEL = {"failed", "cancelled", "expired"}
|
||||
BATCH_CANCEL_RETRIES = 3
|
||||
|
||||
|
||||
def cancel_batch(
|
||||
client: BatchClient, batch_id: str, *, key: str, provider: str | None
|
||||
) -> BatchObject:
|
||||
last = client.cancel_batch(batch_id, key=key, provider=provider)
|
||||
for _ in range(BATCH_CANCEL_RETRIES - 1):
|
||||
match last:
|
||||
case Success(data=data):
|
||||
return data
|
||||
case UnknownApiError(status_code=500):
|
||||
time.sleep(1)
|
||||
last = client.cancel_batch(batch_id, key=key, provider=provider)
|
||||
case _:
|
||||
break
|
||||
return unwrap(last)
|
||||
|
||||
|
||||
def render_jsonl(model: str) -> bytes:
|
||||
|
|
@ -146,7 +167,10 @@ def assert_batch_object(batch: BatchObject) -> None:
|
|||
|
||||
@pytest.mark.parametrize("cap", CAPABILITIES, ids=[c.id for c in CAPABILITIES])
|
||||
def test_batch_lifecycle(
|
||||
cap: Capability, client: BatchClient, resources: ResourceManager
|
||||
cap: Capability,
|
||||
client: BatchClient,
|
||||
resources: ResourceManager,
|
||||
batch_deployments: None,
|
||||
) -> None:
|
||||
key = resources.key()
|
||||
provider = op_provider(cap)
|
||||
|
|
@ -199,11 +223,9 @@ def test_batch_lifecycle(
|
|||
)
|
||||
if pre_cancel.status == "completed":
|
||||
return
|
||||
cancelled = unwrap(client.cancel_batch(batch.id, key=key, provider=provider))
|
||||
cancelled = cancel_batch(client, batch.id, key=key, provider=provider)
|
||||
assert cancelled.id == batch.id
|
||||
assert cancelled.object == "batch"
|
||||
# Vertex cancel is async: the job may still show its pre-cancel status
|
||||
# briefly before transitioning to cancelling/cancelled.
|
||||
valid_post_cancel = {"cancelling", "cancelled"}
|
||||
if cap.provider == "vertex_ai":
|
||||
valid_post_cancel |= CREATED_BATCH_STATUSES
|
||||
|
|
@ -213,7 +235,6 @@ def test_batch_lifecycle(
|
|||
|
||||
if cap.can_list:
|
||||
listed = unwrap(client.list_batches(key=key, provider=provider))
|
||||
# OpenAI includes object="list"; Azure provider list often omits the envelope field.
|
||||
if listed.object is not None:
|
||||
assert listed.object == "list", f"list envelope object={listed.object!r}"
|
||||
match = next((b for b in listed.data if b.id == batch.id), None)
|
||||
|
|
@ -222,7 +243,7 @@ def test_batch_lifecycle(
|
|||
|
||||
|
||||
def test_batch_key_model_access_denied(
|
||||
client: BatchClient, resources: ResourceManager
|
||||
client: BatchClient, resources: ResourceManager, batch_deployments: None
|
||||
) -> None:
|
||||
key = resources.key(models=["openai-batch"])
|
||||
|
||||
|
|
@ -257,7 +278,7 @@ def test_batch_key_model_access_denied(
|
|||
|
||||
|
||||
def test_file_upload_and_delete_outputs(
|
||||
client: BatchClient, resources: ResourceManager
|
||||
client: BatchClient, resources: ResourceManager, batch_deployments: None
|
||||
) -> None:
|
||||
key = resources.key()
|
||||
file = unwrap(
|
||||
|
|
@ -276,14 +297,69 @@ def test_file_upload_and_delete_outputs(
|
|||
assert deleted.deleted is True, "file was not reported deleted"
|
||||
|
||||
|
||||
def test_anthropic_batch_retrieve(client: BatchClient, scoped_key: str) -> None:
|
||||
batch_id = os.environ.get("ANTHROPIC_BATCH_ID")
|
||||
if not batch_id:
|
||||
pytest.skip(
|
||||
"set ANTHROPIC_BATCH_ID to a real anthropic batch id to exercise retrieve"
|
||||
)
|
||||
fetched = unwrap(
|
||||
client.retrieve_batch(batch_id, key=scoped_key, provider="anthropic")
|
||||
def unattributed_rows(rows: list[SpendLogRow]) -> list[SpendLogRow]:
|
||||
"""Spend rows that carry no caller identity (empty api_key).
|
||||
|
||||
Every request the proxy bills is stamped with the calling key. A row with no
|
||||
api_key is one the proxy could not attribute; LIT-3266 is exactly this: the
|
||||
batch rate limiter's internal input-file read ran without the batch's auth
|
||||
metadata, landing a spend row with empty api_key/user. The symptom is not
|
||||
tied to a single call_type, so this catches any unattributed row rather than
|
||||
only a named file-content one.
|
||||
"""
|
||||
return [row for row in rows if not row.api_key]
|
||||
|
||||
|
||||
def test_rate_limited_batch_create_leaves_no_unattributed_spend_row(
|
||||
client: BatchClient, resources: ResourceManager, batch_deployments: None
|
||||
) -> None:
|
||||
"""LIT-3266: creating a batch on a rate-limited key runs the batch rate
|
||||
limiter, which reads the input file to count tokens (the limiter only reads
|
||||
the file when the key has applicable rpm/tpm limits, so an unlimited key
|
||||
hides the path). That internal read must carry the batch's auth metadata;
|
||||
the reported gap was that it did not, spawning a spend-log row with empty
|
||||
api_key/user. Create returning 200 is not a reliable signal (the read error
|
||||
is swallowed), so this asserts the hygiene contract instead: the operation
|
||||
introduces no new unattributed spend row.
|
||||
|
||||
The key sets generous rpm/tpm limits (not a restrictive model allowlist) so
|
||||
the file-read path fires while the batch itself is not blocked.
|
||||
``resources.key()`` cannot set limits, so the key is minted on the gateway
|
||||
directly and its delete deferred.
|
||||
"""
|
||||
user_id = f"e2e-batch-rl-{unique_marker()}"
|
||||
key = client.gateway.generate_key(
|
||||
KeyGenerateBody(models=[], tpm_limit=1_000_000, rpm_limit=1_000, user_id=user_id)
|
||||
)
|
||||
resources.defer(lambda: client.gateway.delete_key(key))
|
||||
|
||||
before = frozenset(
|
||||
row.request_id for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams()))
|
||||
)
|
||||
|
||||
file = unwrap(
|
||||
client.upload_file(
|
||||
content=render_jsonl("gpt-4o-mini"),
|
||||
form=FileUploadForm(purpose="batch"),
|
||||
model="openai-batch",
|
||||
key=key,
|
||||
)
|
||||
)
|
||||
resources.defer(quietly(lambda: client.delete_file(file.id, key=key)))
|
||||
|
||||
created = client.create_batch(body=BatchCreateBody(input_file_id=file.id), key=key)
|
||||
require_successful_call(created)
|
||||
batch = BatchObject.model_validate_json(created.body)
|
||||
resources.defer(quietly(lambda: client.cancel_batch(batch.id, key=key)))
|
||||
|
||||
_ = client.gateway.poll_logs_for_key(key, min_rows=1)
|
||||
|
||||
new_orphans = [
|
||||
row
|
||||
for row in unattributed_rows(client.gateway.spend_logs(SpendLogsParams()))
|
||||
if row.request_id not in before
|
||||
]
|
||||
assert not new_orphans, (
|
||||
"batch create on a rate-limited key left an unattributed spend row "
|
||||
f"(LIT-3266); rows={[(r.request_id, r.call_type, r.model) for r in new_orphans]}"
|
||||
)
|
||||
assert fetched.id == batch_id
|
||||
assert fetched.status
|
||||
|
|
|
|||
|
|
@ -11,6 +11,13 @@ from endpoints_client import EndpointsClient, build_endpoints_client
|
|||
from passthrough_client import PassthroughClient, build_client
|
||||
|
||||
|
||||
def pytest_configure(config: pytest.Config) -> None:
|
||||
config.addinivalue_line(
|
||||
"markers",
|
||||
"covers: registry cell a test covers, e.g. llm.chat_completions.provider.basic.nonstream.works",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def client() -> PassthroughClient:
|
||||
return build_client()
|
||||
|
|
|
|||
|
|
@ -0,0 +1,58 @@
|
|||
"""Live regression net for /chat/completions across the configured providers.
|
||||
|
||||
GH #28991 broke /chat/completions (and /responses) for most models on some
|
||||
releases: a clean 200 came back but with no real completion. A status check
|
||||
alone would not have caught it, so each case here asserts the product promise -
|
||||
a non-empty assistant message and a real model name in the body - across the
|
||||
three providers wired into the gateway config (OpenAI, Anthropic, Gemini). A
|
||||
regression that empties the completion for any provider fails that provider's
|
||||
row here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from models import ChatBody, ChatMessage
|
||||
from passthrough_client import PassthroughClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
CHAT_MODELS: tuple[tuple[str, str], ...] = (
|
||||
("gpt-5.5", "openai"),
|
||||
("claude-haiku-4-5", "anthropic"),
|
||||
("gemini-2.5-flash", "gemini"),
|
||||
)
|
||||
|
||||
|
||||
class TestChatCompletionsRegression:
|
||||
@pytest.mark.parametrize(
|
||||
("model", "route"),
|
||||
CHAT_MODELS,
|
||||
ids=[f"{model}-{route}" for model, route in CHAT_MODELS],
|
||||
)
|
||||
@pytest.mark.covers("llm.chat_completions.provider.basic.nonstream.works", exercised_on=[])
|
||||
def test_chat_returns_real_completion(
|
||||
self, client: PassthroughClient, scoped_key: str, model: str, route: str
|
||||
) -> None:
|
||||
response = unwrap(
|
||||
client.gateway.chat(
|
||||
scoped_key,
|
||||
ChatBody(
|
||||
model=model,
|
||||
messages=[
|
||||
ChatMessage(role="user", content=f"reply with one word {unique_marker()}")
|
||||
],
|
||||
max_tokens=512,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
assert response.model, f"{model} ({route}): response carried no model name: {response}"
|
||||
assert response.choices, f"{model} ({route}): response had no choices: {response}"
|
||||
message = response.choices[0].message
|
||||
assert message is not None and message.content and message.content.strip(), (
|
||||
f"{model} ({route}): 200 with an empty completion (#28991): {response}"
|
||||
)
|
||||
133
tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py
Normal file
133
tests/e2e/llm_translation/test_deepseek_reasoning_e2e.py
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
"""Live e2e: DeepSeek reasoner honors a request to turn reasoning OFF.
|
||||
|
||||
DeepSeek's reasoner defaults thinking ON and surfaces the chain as
|
||||
``message.reasoning_content``. Two documented ways to disable it are
|
||||
``reasoning_effort="none"`` and ``thinking={"type": "disabled"}``. Today the
|
||||
DeepSeek param mapper (``litellm/llms/deepseek/chat/transformation.py``
|
||||
``map_openai_params``) drops both without forwarding any disable signal, so the
|
||||
outbound body carries no ``thinking`` key and DeepSeek keeps thinking on; the
|
||||
response still comes back with ``reasoning_content``. That is the product gap
|
||||
tracked by LIT-3686 / GH #27453.
|
||||
|
||||
The control case proves the model and path work (reasoning is returned when
|
||||
nothing asks to disable it), so the two disable assertions are meaningful. Those
|
||||
two are marked xfail(strict) until the mapper forwards a real disable signal; an
|
||||
xpass then alerts that the fix landed.
|
||||
|
||||
Requires DEEPSEEK_API_KEY on the proxy (tests/e2e/.env). No skip gate: once the
|
||||
proxy is up, a failure here is real, per the suite's hard-fail contract.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody, ThinkingParam
|
||||
from passthrough_client import PassthroughClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
REASONER = "deepseek/deepseek-reasoner"
|
||||
PROMPT = "What is 17 + 26? Answer with just the number."
|
||||
|
||||
|
||||
def _register_reasoner(client: PassthroughClient, resources: ResourceManager) -> str:
|
||||
model = f"e2e-deepseek-reasoner-{unique_marker()}"
|
||||
model_id = client.gateway.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(model=REASONER, api_key="os.environ/DEEPSEEK_API_KEY"),
|
||||
)
|
||||
resources.defer(lambda: client.gateway.delete_model(model_id))
|
||||
return model
|
||||
|
||||
|
||||
def _reasoning_content(response: ChatResponse) -> str | None:
|
||||
assert response.choices, f"reasoner returned no choices: {response}"
|
||||
message = response.choices[0].message
|
||||
assert message is not None, f"reasoner choice has no message: {response}"
|
||||
return message.reasoning_content
|
||||
|
||||
|
||||
class TestDeepSeekReasoningDisable:
|
||||
def test_reasoner_returns_reasoning_by_default(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = _register_reasoner(client, resources)
|
||||
key = resources.key()
|
||||
|
||||
response = unwrap(
|
||||
client.gateway.chat(
|
||||
key,
|
||||
ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=PROMPT)],
|
||||
max_tokens=64,
|
||||
),
|
||||
)
|
||||
)
|
||||
reasoning = _reasoning_content(response)
|
||||
assert reasoning, (
|
||||
"control case: deepseek-reasoner returned no reasoning_content with no "
|
||||
f"disable param, so the disable assertions below can't be trusted: {response}"
|
||||
)
|
||||
|
||||
@pytest.mark.xfail(
|
||||
strict=True,
|
||||
reason=(
|
||||
"LIT-3686 / GH #27453: DeepSeek reasoning_effort='none' and "
|
||||
"thinking type='disabled' are silently dropped; reasoning not disabled"
|
||||
),
|
||||
)
|
||||
def test_reasoning_effort_none_disables_reasoning(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = _register_reasoner(client, resources)
|
||||
key = resources.key()
|
||||
|
||||
response = unwrap(
|
||||
client.gateway.chat(
|
||||
key,
|
||||
ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=PROMPT)],
|
||||
max_tokens=64,
|
||||
reasoning_effort="none",
|
||||
),
|
||||
)
|
||||
)
|
||||
assert not _reasoning_content(response), (
|
||||
"reasoning_effort='none' must disable reasoning, but reasoning_content "
|
||||
f"is still present: {response}"
|
||||
)
|
||||
|
||||
@pytest.mark.xfail(
|
||||
strict=True,
|
||||
reason=(
|
||||
"LIT-3686 / GH #27453: DeepSeek reasoning_effort='none' and "
|
||||
"thinking type='disabled' are silently dropped; reasoning not disabled"
|
||||
),
|
||||
)
|
||||
def test_thinking_disabled_disables_reasoning(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = _register_reasoner(client, resources)
|
||||
key = resources.key()
|
||||
|
||||
response = unwrap(
|
||||
client.gateway.chat(
|
||||
key,
|
||||
ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=PROMPT)],
|
||||
max_tokens=64,
|
||||
thinking=ThinkingParam(type="disabled"),
|
||||
),
|
||||
)
|
||||
)
|
||||
assert not _reasoning_content(response), (
|
||||
"thinking={'type': 'disabled'} must disable reasoning, but "
|
||||
f"reasoning_content is still present: {response}"
|
||||
)
|
||||
148
tests/e2e/llm_translation/test_provider_features_e2e.py
Normal file
148
tests/e2e/llm_translation/test_provider_features_e2e.py
Normal file
|
|
@ -0,0 +1,148 @@
|
|||
"""Live e2e for model-specific request features: service_tier and prompt caching.
|
||||
|
||||
Each case asserts the feature took effect, not just a 200.
|
||||
|
||||
service_tier is an OpenAI concept. The proxy forwards it and the provider echoes
|
||||
the tier back on the response, so sending a non-default tier ("flex") and reading
|
||||
it back off ``service_tier`` proves the param was honored end to end; litellm's own
|
||||
default injection would report "default", so a "flex" echo can only come from the
|
||||
request being forwarded. Bedrock and Vertex do not accept service_tier, so that
|
||||
cell is OpenAI-only by design.
|
||||
|
||||
Prompt caching is asserted through provider prompt-cache usage tokens. The
|
||||
deterministic path is explicit ``cache_control`` on an Anthropic-family model
|
||||
(here Bedrock's Claude): a large cacheable prefix is sent twice and the second
|
||||
call must report ``cache_read_input_tokens > 0``. OpenAI and Gemini only offer
|
||||
implicit automatic caching, which does not deterministically produce a cache read
|
||||
within a test window (verified: repeated >3k-token prompts kept
|
||||
``prompt_tokens_details.cached_tokens`` at 0), so those caching cells are out of
|
||||
scope here and covered only by the explicit-cache-control Bedrock case.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody
|
||||
from passthrough_client import PassthroughClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
SERVICE_TIER = "flex"
|
||||
CACHE_MIN_READ_TOKENS = 1
|
||||
|
||||
|
||||
class CacheControl(BaseModel):
|
||||
type: str = "ephemeral"
|
||||
|
||||
|
||||
class CacheTextBlock(BaseModel):
|
||||
type: str = "text"
|
||||
text: str
|
||||
cache_control: CacheControl | None = None
|
||||
|
||||
|
||||
class RichMessage(BaseModel):
|
||||
role: str
|
||||
content: list[CacheTextBlock]
|
||||
|
||||
|
||||
class CacheChatBody(BaseModel):
|
||||
model: str
|
||||
messages: list[RichMessage]
|
||||
max_tokens: int
|
||||
|
||||
|
||||
def cacheable_prefix() -> str:
|
||||
return (
|
||||
"You are a policy compliance auditor. The following corpus is the immutable "
|
||||
"reference the assistant must consult on every turn. "
|
||||
) + ("Clause: obey all safety, formatting, and citation rules exactly. " * 400)
|
||||
|
||||
|
||||
def post_chat(client: PassthroughClient, key: str, body: BaseModel) -> ChatResponse:
|
||||
return unwrap(
|
||||
client.gateway.transport.post(
|
||||
"/chat/completions",
|
||||
headers=client.gateway.transport.bearer(key),
|
||||
json=body,
|
||||
response_type=ChatResponse,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
class TestServiceTier:
|
||||
@pytest.mark.covers("llm.chat_completions.openai.service_tier.works", exercised_on=[])
|
||||
def test_openai_service_tier_is_echoed(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = f"e2e-service-tier-{unique_marker()}"
|
||||
model_id = client.gateway.create_model(
|
||||
model, LiteLLMParamsBody(model="openai/gpt-5.5", api_key="os.environ/OPENAI_API_KEY")
|
||||
)
|
||||
resources.defer(lambda: client.gateway.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
||||
response = unwrap(
|
||||
client.gateway.chat(
|
||||
key,
|
||||
ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content="reply with one word")],
|
||||
max_tokens=64,
|
||||
service_tier=SERVICE_TIER,
|
||||
),
|
||||
)
|
||||
)
|
||||
assert response.service_tier == SERVICE_TIER, (
|
||||
f"service_tier not honored: sent {SERVICE_TIER!r}, response reported "
|
||||
f"{response.service_tier!r} ({response})"
|
||||
)
|
||||
|
||||
|
||||
class TestPromptCaching:
|
||||
@pytest.mark.covers(
|
||||
"llm.chat_completions.bedrock_converse.prompt_cache_5m.nonstream.cache_hit", exercised_on=[]
|
||||
)
|
||||
def test_bedrock_cache_control_produces_cache_read(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
model = f"e2e-bedrock-cache-{unique_marker()}"
|
||||
model_id = client.gateway.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
||||
aws_region_name="us-east-1",
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: client.gateway.delete_model(model_id))
|
||||
key = resources.key()
|
||||
|
||||
body = CacheChatBody(
|
||||
model=model,
|
||||
max_tokens=32,
|
||||
messages=[
|
||||
RichMessage(
|
||||
role="user",
|
||||
content=[
|
||||
CacheTextBlock(text=cacheable_prefix(), cache_control=CacheControl()),
|
||||
CacheTextBlock(text="Answer in one word: acknowledged?"),
|
||||
],
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
first = post_chat(client, key, body)
|
||||
assert first.usage is not None, f"first call reported no usage: {first}"
|
||||
|
||||
second = post_chat(client, key, body)
|
||||
assert second.usage is not None, f"second call reported no usage: {second}"
|
||||
cache_read = second.usage.cache_read_input_tokens
|
||||
assert cache_read is not None and cache_read >= CACHE_MIN_READ_TOKENS, (
|
||||
"second identical request did not read the prompt cache: "
|
||||
f"cache_read_input_tokens={cache_read!r} (usage={second.usage})"
|
||||
)
|
||||
37
tests/e2e/logging/conftest.py
Normal file
37
tests/e2e/logging/conftest.py
Normal file
|
|
@ -0,0 +1,37 @@
|
|||
"""Fixtures for the Datadog logging suite.
|
||||
|
||||
These tests drive the Datadog batch-send path (#25663) directly against the real
|
||||
Datadog logs intake with synthetic events - no LLM calls, no proxy, no log
|
||||
read-back - so they need only the shipping credentials DD_API_KEY + DD_SITE
|
||||
(DD_SERVICE is an optional tag). No Datadog Application key is required, and they
|
||||
skip when the shipping credentials are absent from the environment.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from logging_client import LoggingClient, build_logging_client
|
||||
|
||||
|
||||
def pytest_configure(config: pytest.Config) -> None:
|
||||
config.addinivalue_line(
|
||||
"markers",
|
||||
"covers: registry cell a test covers, e.g. logging.datadog.success.writes_object",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def client() -> LoggingClient:
|
||||
"""The logging suite's client: holds the shared Gateway so `resources` /
|
||||
`scoped_key` clean up keys, and adds `/metrics` scraping."""
|
||||
return build_logging_client()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def datadog_creds() -> None:
|
||||
"""Gate the suite on the Datadog shipping credentials. The DataDogLogger is built
|
||||
inside each async test, not here, because its __init__ schedules a periodic-flush
|
||||
task via asyncio.create_task and so needs a running event loop."""
|
||||
if not (os.getenv("DD_API_KEY") and os.getenv("DD_SITE")):
|
||||
pytest.skip("set DD_API_KEY and DD_SITE to run the Datadog logging suite")
|
||||
48
tests/e2e/logging/logging_client.py
Normal file
48
tests/e2e/logging/logging_client.py
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
"""Client for the logging e2e suite: drive traffic and scrape the proxy's
|
||||
Prometheus ``/metrics`` endpoint.
|
||||
|
||||
Holds the shared Gateway so the ``resources`` fixture cleans up keys it creates.
|
||||
``/metrics`` is exposed as plaintext (not a typed JSON body), so scraping goes
|
||||
through ``transport.probe`` and returns the raw exposition text for a Prometheus
|
||||
parser to read.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from e2e_gateway import Gateway, build_gateway
|
||||
from e2e_http import NoBody, unwrap
|
||||
from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class LoggingClient:
|
||||
gateway: Gateway
|
||||
|
||||
def key_with_alias(self, alias: str, *, models: list[str]) -> str:
|
||||
return self.gateway.generate_key(
|
||||
KeyGenerateBody(key_alias=alias, models=models, user_id=f"e2e-{alias}")
|
||||
)
|
||||
|
||||
def delete_key(self, key: str) -> None:
|
||||
self.gateway.delete_key(key)
|
||||
|
||||
def chat(self, key: str, model: str, text: str) -> ChatResponse:
|
||||
return unwrap(
|
||||
self.gateway.chat(
|
||||
key,
|
||||
ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=text)],
|
||||
max_tokens=64,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
def scrape_metrics(self) -> str:
|
||||
return self.gateway.probe("/metrics", params=NoBody()).body
|
||||
|
||||
|
||||
def build_logging_client() -> LoggingClient:
|
||||
return LoggingClient(gateway=build_gateway())
|
||||
70
tests/e2e/logging/test_prometheus_cardinality_e2e.py
Normal file
70
tests/e2e/logging/test_prometheus_cardinality_e2e.py
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
"""Live e2e: Prometheus request metrics grow one series per virtual key.
|
||||
|
||||
The proxy exposes ``/metrics`` (prometheus is in the callbacks and
|
||||
``require_auth_for_metrics_endpoint`` is off in the e2e config). The counter
|
||||
``litellm_requests_metric_total`` carries an ``api_key_alias`` label, so driving
|
||||
traffic through keys with distinct aliases must produce a distinct labeled series
|
||||
per alias. This is the per-key cardinality contract: a regression that stops
|
||||
stamping ``api_key_alias`` (or collapses every key onto one series) would drop
|
||||
the aliases and fail here.
|
||||
|
||||
Scraping goes through ``transport.probe`` (raw text) and is parsed with
|
||||
prometheus_client; the metric is eventually consistent (it increments on the
|
||||
success-logging callback), so the scrape polls to a deadline.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
|
||||
import pytest
|
||||
from prometheus_client.parser import text_string_to_metric_families
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from lifecycle import ResourceManager
|
||||
from logging_client import LoggingClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
DRIVER_MODEL = "gemini-2.5-flash"
|
||||
REQUESTS_METRIC = "litellm_requests_metric_total"
|
||||
ALIAS_LABEL = "api_key_alias"
|
||||
DISTINCT_KEYS = 3
|
||||
|
||||
|
||||
def _aliases_in_metric(exposition: str, metric: str, label: str) -> frozenset[str]:
|
||||
"""The set of ``label`` values present on ``metric`` samples in a scrape."""
|
||||
return frozenset(
|
||||
sample.labels[label]
|
||||
for family in text_string_to_metric_families(exposition)
|
||||
for sample in family.samples
|
||||
if sample.name == metric and label in sample.labels
|
||||
)
|
||||
|
||||
|
||||
class TestPrometheusPerKeyCardinality:
|
||||
@pytest.mark.covers("logging.prometheus.success.exports_metric", exercised_on=[])
|
||||
def test_distinct_key_aliases_produce_distinct_series(
|
||||
self, client: LoggingClient, resources: ResourceManager
|
||||
) -> None:
|
||||
aliases = tuple(f"e2e-prom-{unique_marker()}" for _ in range(DISTINCT_KEYS))
|
||||
for alias in aliases:
|
||||
key = client.key_with_alias(alias, models=[DRIVER_MODEL])
|
||||
resources.defer(lambda k=key: client.delete_key(k))
|
||||
response = client.chat(key, DRIVER_MODEL, f"reply with one word {alias}")
|
||||
assert response.model, f"driver call for {alias} returned no model: {response}"
|
||||
|
||||
wanted = frozenset(aliases)
|
||||
deadline = time.monotonic() + client.gateway.poll_timeout
|
||||
seen: frozenset[str] = frozenset()
|
||||
while time.monotonic() < deadline:
|
||||
seen = _aliases_in_metric(client.scrape_metrics(), REQUESTS_METRIC, ALIAS_LABEL)
|
||||
if wanted <= seen:
|
||||
break
|
||||
time.sleep(client.gateway.poll_interval)
|
||||
|
||||
missing = wanted - seen
|
||||
assert not missing, (
|
||||
f"{REQUESTS_METRIC} is missing a per-key series for aliases {sorted(missing)}; "
|
||||
f"each distinct {ALIAS_LABEL} must grow its own series"
|
||||
)
|
||||
|
|
@ -6,6 +6,8 @@ response validates without mirroring every proxy field. No untyped dicts.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, RootModel
|
||||
|
||||
# ---------- keys ----------
|
||||
|
|
@ -30,6 +32,7 @@ class KeyGenerateBody(BaseModel):
|
|||
user_id: str | None = None
|
||||
team_id: str | None = None
|
||||
budget_id: str | None = None
|
||||
key_alias: str | None = None
|
||||
model_max_budget: dict[str, ModelBudgetEntry] | None = None
|
||||
budget_fallbacks: dict[str, list[str]] | None = None
|
||||
budget_limits: list[BudgetWindow] | None = None
|
||||
|
|
@ -88,6 +91,16 @@ class ChatMessage(BaseModel):
|
|||
content: str
|
||||
|
||||
|
||||
class ThinkingParam(BaseModel):
|
||||
"""Extended-thinking control shared by Anthropic and DeepSeek reasoner models.
|
||||
DeepSeek accepts only ``type`` (enabled/disabled) and ignores budget_tokens;
|
||||
Anthropic also honors budget_tokens. Sending ``type="disabled"`` is the
|
||||
product-facing way a caller turns reasoning off (LIT-3686 / GH #27453)."""
|
||||
|
||||
type: Literal["enabled", "disabled"]
|
||||
budget_tokens: int | None = None
|
||||
|
||||
|
||||
class ChatBody(BaseModel):
|
||||
model: str
|
||||
messages: list[ChatMessage]
|
||||
|
|
@ -95,6 +108,9 @@ class ChatBody(BaseModel):
|
|||
max_tokens: int | None = None
|
||||
user: str | None = None
|
||||
metadata: ChatMetadata | None = None
|
||||
reasoning_effort: str | None = None
|
||||
thinking: ThinkingParam | None = None
|
||||
service_tier: str | None = None
|
||||
|
||||
|
||||
class AnthropicMessagesBody(BaseModel):
|
||||
|
|
@ -105,16 +121,24 @@ class AnthropicMessagesBody(BaseModel):
|
|||
|
||||
class OutMessage(BaseModel):
|
||||
content: str | None = None
|
||||
reasoning_content: str | None = None
|
||||
|
||||
|
||||
class ChatChoice(BaseModel):
|
||||
message: OutMessage | None = None
|
||||
|
||||
|
||||
class PromptTokensDetails(BaseModel):
|
||||
cached_tokens: int | None = None
|
||||
|
||||
|
||||
class Usage(BaseModel):
|
||||
prompt_tokens: int | None = None
|
||||
completion_tokens: int | None = None
|
||||
total_tokens: int | None = None
|
||||
cache_read_input_tokens: int | None = None
|
||||
cache_creation_input_tokens: int | None = None
|
||||
prompt_tokens_details: PromptTokensDetails | None = None
|
||||
|
||||
|
||||
class ChatResponse(BaseModel):
|
||||
|
|
@ -122,6 +146,7 @@ class ChatResponse(BaseModel):
|
|||
model: str | None = None
|
||||
choices: list[ChatChoice] = []
|
||||
usage: Usage | None = None
|
||||
service_tier: str | None = None
|
||||
|
||||
|
||||
class EmbedBody(BaseModel):
|
||||
|
|
@ -166,6 +191,7 @@ class OcrResponse(BaseModel):
|
|||
|
||||
class SpendLogRow(BaseModel):
|
||||
request_id: str | None = None
|
||||
api_key: str | None = None
|
||||
model: str | None = None
|
||||
spend: float | None = None
|
||||
status: str | None = None
|
||||
|
|
@ -287,6 +313,32 @@ class ModelInfoResponse(BaseModel):
|
|||
data: list[ModelInfoEntry] = []
|
||||
|
||||
|
||||
class FileEntry(BaseModel):
|
||||
id: str
|
||||
|
||||
|
||||
class FileListResponse(BaseModel):
|
||||
"""GET /files answer. `data` is required on purpose: a 200 whose body lacks
|
||||
the OpenAI-format file list must fail validation, not pass vacuously."""
|
||||
|
||||
data: list[FileEntry]
|
||||
|
||||
|
||||
class FineTuningJobsParams(BaseModel):
|
||||
custom_llm_provider: Literal["openai", "azure"]
|
||||
|
||||
|
||||
class FineTuningJobEntry(BaseModel):
|
||||
id: str
|
||||
|
||||
|
||||
class FineTuningJobsResponse(BaseModel):
|
||||
"""GET /fine_tuning/jobs answer; `data` required for the same reason as
|
||||
FileListResponse."""
|
||||
|
||||
data: list[FineTuningJobEntry]
|
||||
|
||||
|
||||
# ---------- model management ----------
|
||||
|
||||
|
||||
|
|
@ -301,12 +353,23 @@ class LiteLLMParamsBody(BaseModel):
|
|||
api_key: str | None = None
|
||||
api_base: str | None = None
|
||||
api_version: str | None = None
|
||||
aws_region_name: str | None = None
|
||||
vertex_project: str | None = None
|
||||
vertex_location: str | None = None
|
||||
vertex_credentials: str | None = None
|
||||
bucket_name: str | None = None
|
||||
s3_bucket_name: str | None = None
|
||||
s3_region_name: str | None = None
|
||||
s3_access_key_id: str | None = None
|
||||
s3_secret_access_key: str | None = None
|
||||
aws_batch_role_arn: str | None = None
|
||||
input_cost_per_token: float | None = None
|
||||
output_cost_per_token: float | None = None
|
||||
|
||||
|
||||
class ModelInfoBody(BaseModel):
|
||||
id: str
|
||||
mode: Literal["batch", "realtime", "image_generation"] | None = None
|
||||
|
||||
|
||||
class ModelNewBody(BaseModel):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue