mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
* test(e2e): add failing reproducers for two open gateway bugs
Both tests assert the behavior a customer expects and both are red today. They
are reproducers, not regressions: the product is wrong, not the tests.
Native passthrough returns almost none of the operational headers the managed
route does. A /gemini/ generateContent call comes back with three x-litellm-*
headers and no x-ratelimit-* at all, against sixteen and four on
/v1beta/models/{m}:generateContent for the same prompt, and critically it omits
x-litellm-response-cost. Customers front provider-native traffic through this
route and read those headers to reconcile spend and pace themselves, so native
traffic is currently invisible to the tooling that covers every other route.
/budget/update rejects any model_max_budget with a 500. The reported symptom was
model ids containing dots, and that reproduces (prisma raises "Unexpected
`-5.2[FloatValue]` Expected `:`" because the key is interpolated into a GraphQL
query unquoted, so glm-5.2 lexes as an identifier followed by a float), but the
plain name gpt4o fails too, on a separate "model_max_budget should be of any of
the following types: Json" type mismatch at budget_management_endpoints.py:173.
Omitting the field returns 200. The test drives both names so the failure says
whether per-model budgets are broken outright or only for punctuated ids; today
it stops on the plain name, which is the wider bug.
* test(e2e): add reproducer for unenforced end-user per-model rate limits
model_max_budget accepts an rpm_limit alongside the spend cap, and /budget/new
stores it: the create response echoes {"gemini-2.5-flash": {"rpm_limit": 1,
"max_budget": 100.0, "budget_duration": "1d"}}. Attach that budget to an end
user, drive three calls as that user, and all three return 200. The limit is
accepted, persisted, and then ignored.
The same shape already works when the budget hangs off a key, which is what
makes this quietly dangerous: the API gives every indication the cap is in
force. A customer using it to hold one end user to a slow rate on a shared key
gets no throttling at all.
Harness additions this needs: ModelBudgetEntry carries the rpm_limit/tpm_limit
the route already accepts, BudgetNewBody and create_budget carry
model_max_budget, and create_customer can attach an existing budget_id rather
than only an inline max_budget.
Red today, for the reason in the assertion message.
* test(e2e): tighten model_max_budget reproducers and drop in-loop closure
Trim the reproducer docstrings to the contract they assert, keeping the
failure messages that document each red-by-design bug. Replace the nested
per-model closure in the /budget/update test with a module-level predicate
and a per-model helper so nothing closes over a loop variable, and fix the
import order the merge left unsorted.
* test(e2e): skip the three reproducers while their gateway bugs stay open
The passthrough header contract, /budget/update model_max_budget, and
end-user per-model rpm enforcement reproducers all still fail against
staging by design. Skip each with the product gap named so the combined
suite can gate merges on green while the collector keeps reporting the
cells as uncovered.
* test(e2e): validate model budget response contracts
* refactor(e2e): unify model budget schema
* refactor(e2e): reuse shared model budget type
212 lines
7.9 KiB
Python
212 lines
7.9 KiB
Python
"""Live e2e for LLM-translation passthrough endpoints.
|
|
|
|
Each test sends a NATIVE provider request through the proxy's passthrough route
|
|
and verifies the proxy still logged a costed SpendLogs row
|
|
(call_type="pass_through_endpoint"), correlated by the x-litellm-call-id header.
|
|
|
|
Covered: gemini ("gemini-2.5-flash") + anthropic ("claude-haiku-4-5"), streaming +
|
|
non-streaming, plus native tool calls. See LLM_TRANSLATION_COVERAGE_MATRIX.md.
|
|
|
|
A passthrough call returning non-2xx fails hard (never a skip); once it returns
|
|
2xx, a missing or zero-cost SpendLogs row fails too.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from e2e_config import unique_marker
|
|
from e2e_http import StreamingResponse, require_successful_call
|
|
from lifecycle import ResourceManager
|
|
from models import KeyGenerateBody, SpendLogRow
|
|
from passthrough_client import (
|
|
AnthropicTool,
|
|
GeminiFunctionDeclaration,
|
|
GeminiTool,
|
|
JsonSchema,
|
|
JsonSchemaProperty,
|
|
PassthroughClient,
|
|
)
|
|
|
|
pytestmark = pytest.mark.e2e
|
|
|
|
|
|
def _fetch_cost_breakdown(client: PassthroughClient, result: StreamingResponse) -> SpendLogRow:
|
|
"""The passthrough call's logged row, polled until it carries a cost.
|
|
|
|
Asserts (not skips) that a 2xx passthrough call produced a costed row - the
|
|
whole point of passthrough spend tracking.
|
|
"""
|
|
assert result.call_id, "passthrough response had no x-litellm-call-id header"
|
|
rows = client.proxy.poll_logs_for_request_id(
|
|
result.call_id,
|
|
predicate=lambda rs: (rs[0].spend or 0) > 0,
|
|
)
|
|
assert rows, f"no SpendLogs row for passthrough call_id {result.call_id}"
|
|
row = rows[0]
|
|
assert row.call_type == "pass_through_endpoint"
|
|
assert (row.spend or 0) > 0, f"passthrough call was not costed: {row}"
|
|
assert row.status == "success"
|
|
return row
|
|
|
|
|
|
# ---- Gemini passthrough ------------------------------------------------
|
|
|
|
|
|
def test_gemini_passthrough_nonstreaming_logs_cost(
|
|
client: PassthroughClient, scoped_key: str
|
|
) -> None:
|
|
tag = f"e2e-passthrough-{unique_marker()}"
|
|
result = client.gemini_generate(
|
|
scoped_key, "gemini-2.5-flash", "Say hello in one word", tags=[tag, "gemini"]
|
|
)
|
|
require_successful_call(result)
|
|
|
|
row = _fetch_cost_breakdown(client, result)
|
|
assert row.custom_llm_provider == "gemini"
|
|
assert "gemini" in (row.model or "")
|
|
assert tag in (row.request_tags or []), f"tags not logged: {row.request_tags}"
|
|
|
|
|
|
@pytest.mark.skip(reason="stage red: product gap, native passthrough returns no x-litellm-response-cost or x-ratelimit-* headers")
|
|
def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route(
|
|
client: PassthroughClient, scoped_key: str
|
|
) -> None:
|
|
"""Native /gemini/ passthrough must return the same operational headers as
|
|
/chat/completions: x-litellm-response-cost so the call reconciles against
|
|
spend, and x-ratelimit-* so a client can pace itself. It returns neither
|
|
today, which makes native traffic invisible to the same tooling.
|
|
"""
|
|
result = client.gemini_generate(
|
|
scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}"
|
|
)
|
|
require_successful_call(result)
|
|
|
|
assert result.call_id, "passthrough must stamp x-litellm-call-id"
|
|
assert result.response_cost is not None, (
|
|
"passthrough generateContent returned no x-litellm-response-cost header, so a "
|
|
"native call cannot be reconciled against spend the way /chat/completions can"
|
|
)
|
|
assert result.response_cost > 0, (
|
|
f"x-litellm-response-cost must be a real cost, got {result.response_cost}"
|
|
)
|
|
|
|
pacing = tuple(name for name in result.headers if name.startswith("x-ratelimit-"))
|
|
assert pacing, (
|
|
"passthrough generateContent returned no x-ratelimit-* headers, so a client "
|
|
f"cannot pace itself; headers present were {sorted(result.headers)}"
|
|
)
|
|
|
|
|
|
def test_gemini_passthrough_streaming_logs_cost(
|
|
client: PassthroughClient, scoped_key: str
|
|
) -> None:
|
|
result = client.gemini_stream(scoped_key, "gemini-2.5-flash", "Count to five")
|
|
require_successful_call(result)
|
|
assert result.chunks > 0, "streaming passthrough produced no events"
|
|
|
|
row = _fetch_cost_breakdown(client, result)
|
|
assert row.custom_llm_provider == "gemini"
|
|
|
|
|
|
def test_gemini_passthrough_tool_call_logs_cost(
|
|
client: PassthroughClient, scoped_key: str
|
|
) -> None:
|
|
result = client.gemini_generate(
|
|
scoped_key,
|
|
"gemini-2.5-flash",
|
|
"What is the weather in Paris? Use the get_weather tool.",
|
|
tools=[
|
|
GeminiTool(
|
|
function_declarations=[
|
|
GeminiFunctionDeclaration(
|
|
name="get_weather",
|
|
description="Get the weather for a city",
|
|
parameters=JsonSchema(
|
|
type="object",
|
|
properties={"city": JsonSchemaProperty(type="string")},
|
|
required=["city"],
|
|
),
|
|
)
|
|
]
|
|
)
|
|
],
|
|
)
|
|
require_successful_call(result)
|
|
assert "functionCall" in result.body, "gemini did not emit a tool call"
|
|
|
|
row = _fetch_cost_breakdown(client, result)
|
|
assert row.custom_llm_provider == "gemini"
|
|
|
|
|
|
# ---- Anthropic passthrough ---------------------------------------------
|
|
|
|
|
|
def test_anthropic_passthrough_nonstreaming_logs_cost(
|
|
client: PassthroughClient, scoped_key: str
|
|
) -> None:
|
|
result = client.anthropic_message(scoped_key, "claude-haiku-4-5", "Say hello")
|
|
require_successful_call(result)
|
|
|
|
row = _fetch_cost_breakdown(client, result)
|
|
assert row.custom_llm_provider == "anthropic"
|
|
assert "claude" in (row.model or "")
|
|
|
|
|
|
def test_anthropic_passthrough_streaming_logs_cost(
|
|
client: PassthroughClient, scoped_key: str
|
|
) -> None:
|
|
result = client.anthropic_message(
|
|
scoped_key, "claude-haiku-4-5", "Count to five", stream=True
|
|
)
|
|
require_successful_call(result)
|
|
assert result.chunks > 0, "streaming passthrough produced no events"
|
|
|
|
row = _fetch_cost_breakdown(client, result)
|
|
assert row.custom_llm_provider == "anthropic"
|
|
|
|
|
|
def test_anthropic_passthrough_tool_call_logs_cost(
|
|
client: PassthroughClient, scoped_key: str
|
|
) -> None:
|
|
result = client.anthropic_message(
|
|
scoped_key,
|
|
"claude-haiku-4-5",
|
|
"What is the weather in Paris? Use the get_weather tool.",
|
|
tools=[
|
|
AnthropicTool(
|
|
name="get_weather",
|
|
description="Get the weather for a city",
|
|
input_schema=JsonSchema(
|
|
type="object",
|
|
properties={"city": JsonSchemaProperty(type="string")},
|
|
required=["city"],
|
|
),
|
|
)
|
|
],
|
|
)
|
|
require_successful_call(result)
|
|
assert "tool_use" in result.body, "anthropic did not emit a tool call"
|
|
|
|
row = _fetch_cost_breakdown(client, result)
|
|
assert row.custom_llm_provider == "anthropic"
|
|
|
|
|
|
class TestPassthroughModelAllowlist:
|
|
"""A passthrough route must honor the calling key's model allow-list.
|
|
|
|
The customer fronts native provider calls through the proxy with custom auth,
|
|
so a key scoped to one model must not reach a different model just because the
|
|
request goes through the passthrough route rather than /chat/completions.
|
|
"""
|
|
|
|
@pytest.mark.covers("other.auth.passthrough.model_allowlist_enforced")
|
|
def test_passthrough_denies_model_outside_key_allowlist(
|
|
self, client: PassthroughClient, resources: ResourceManager
|
|
) -> None:
|
|
key = client.proxy.generate_key(KeyGenerateBody(models=["gemini-2.5-flash"]))
|
|
resources.defer(lambda: client.proxy.delete_key(key))
|
|
|
|
result = client.anthropic_message(key, "claude-haiku-4-5", f"say hi {unique_marker()}")
|
|
assert result.status_code == 403, (
|
|
"a key restricted to gemini-2.5-flash must be denied a claude passthrough call, "
|
|
f"got {result.status_code}: {result.body[:300]}"
|
|
)
|