test(e2e): add failing reproducers for two open gateway bugs

Both tests assert the behavior a customer expects and both are red today. They
are reproducers, not regressions: the product is wrong, not the tests.

Native passthrough returns almost none of the operational headers the managed
route does. A /gemini/ generateContent call comes back with three x-litellm-*
headers and no x-ratelimit-* at all, against sixteen and four on
/v1beta/models/{m}:generateContent for the same prompt, and critically it omits
x-litellm-response-cost. Customers front provider-native traffic through this
route and read those headers to reconcile spend and pace themselves, so native
traffic is currently invisible to the tooling that covers every other route.

/budget/update rejects any model_max_budget with a 500. The reported symptom was
model ids containing dots, and that reproduces (prisma raises "Unexpected
`-5.2[FloatValue]` Expected `:`" because the key is interpolated into a GraphQL
query unquoted, so glm-5.2 lexes as an identifier followed by a float), but the
plain name gpt4o fails too, on a separate "model_max_budget should be of any of
the following types: Json" type mismatch at budget_management_endpoints.py:173.
Omitting the field returns 200. The test drives both names so the failure says
whether per-model budgets are broken outright or only for punctuated ids; today
it stops on the plain name, which is the wider bug.
This commit is contained in:
mubashir1osmani 2026-07-25 14:01:41 -07:00
parent a10365e84d
commit 1eaf410989
3 changed files with 92 additions and 2 deletions

View file

@ -54,6 +54,7 @@
- {id: mgmt.access_group.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "model_access_group_management_endpoints.py:600", rationale: "Access group membership query"}
- {id: mgmt.mcp_server.register.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "mcp_management_endpoints.py:880", rationale: "MCP server registration"}
- {id: mgmt.mcp_server.approve.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "mcp_management_endpoints.py:1200", rationale: "Admin approval persists"}
- {id: mgmt.budget.update.accepts_model_max_budget, module: mgmt, tier: P1, surface: api, assertions: [accepts_model_max_budget], source: "budget_management_endpoints.py:173", fail_before_fix: proven, rationale: "Per-model caps must be settable on an existing budget; model ids routinely carry dots and hyphens and the route must accept both"}
- {id: mgmt.budget.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:155", rationale: "Limit changes apply"}
- {id: mgmt.budget.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:280", rationale: "Clears limits"}
- {id: mgmt.budget.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "budget_management_endpoints.py:215", rationale: "Budget enumeration"}

View file

@ -66,6 +66,38 @@ def test_gemini_passthrough_nonstreaming_logs_cost(
assert tag in (row.request_tags or []), f"tags not logged: {row.request_tags}"
def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route(
client: PassthroughClient, scoped_key: str
) -> None:
"""A native passthrough call must still be costable and pace-able by the client.
Customers front provider-native traffic through /gemini/ and read the same
operational headers they get on /chat/completions: the response cost, so the
call reconciles against spend, and the x-ratelimit-* pacing headers, so a
client knows how much budget it has left. The passthrough route currently
returns neither, which makes native traffic invisible to the same tooling.
"""
result = client.gemini_generate(
scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}"
)
require_successful_call(result)
assert result.call_id, "passthrough must stamp x-litellm-call-id"
assert result.response_cost is not None, (
"passthrough generateContent returned no x-litellm-response-cost header, so a "
"native call cannot be reconciled against spend the way /chat/completions can"
)
assert result.response_cost > 0, (
f"x-litellm-response-cost must be a real cost, got {result.response_cost}"
)
pacing = tuple(name for name in result.headers if name.startswith("x-ratelimit-"))
assert pacing, (
"passthrough generateContent returned no x-ratelimit-* headers, so a client "
f"cannot pace itself; headers present were {sorted(result.headers)}"
)
def test_gemini_passthrough_streaming_logs_cost(
client: PassthroughClient, scoped_key: str
) -> None:

View file

@ -22,7 +22,7 @@ import pytest
from pydantic import BaseModel, RootModel
from e2e_config import unique_marker
from e2e_http import NoBody, unwrap
from e2e_http import NoBody, is_ok, unwrap
from lifecycle import ResourceManager
from management_client import ManagementClient
from models import KeyGenerateBody, OrgInfoParams, OrgNewBody, UserNewBody
@ -53,9 +53,15 @@ class BudgetNewResponse(BaseModel):
budget_id: str
class ModelBudgetEntry(BaseModel):
budget_limit: float
time_period: str
class BudgetUpdateBody(BaseModel):
budget_id: str
max_budget: float
max_budget: float | None = None
model_max_budget: dict[str, ModelBudgetEntry] | None = None
class BudgetInfoBody(BaseModel):
@ -66,6 +72,7 @@ class BudgetRow(BaseModel):
budget_id: str | None = None
max_budget: float | None = None
soft_budget: float | None = None
model_max_budget: dict[str, ModelBudgetEntry] | None = None
class BudgetInfoResponse(RootModel[list[BudgetRow]]):
@ -148,6 +155,56 @@ class TestBudgetManagement:
f"/budget/list never included the created budget {budget_id}",
)
@pytest.mark.covers("mgmt.budget.update.accepts_model_max_budget")
def test_update_accepts_per_model_budgets_including_punctuated_names(
self, client: ManagementClient, resources: ResourceManager
) -> None:
"""Per-model caps must be settable on an existing budget.
`model_max_budget` is how a customer caps spend per model on a shared
budget, and model ids routinely carry dots and hyphens (`glm-5.2`). The
route has to accept both, and the plain name is included so a failure
says whether per-model budgets are broken outright or only for
punctuated ids.
"""
for model_name in ("gpt4o", "glm-5.2"):
budget_id = _create_budget(
client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET)
)
result = client.proxy.transport.post(
"/budget/update",
headers=client.proxy.transport.master,
json=BudgetUpdateBody(
budget_id=budget_id,
model_max_budget={
model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d")
},
),
response_type=NoBody,
)
assert is_ok(result), (
f"/budget/update rejected a per-model budget for {model_name!r}: {result}; "
f"a customer cannot cap spend per model on an existing budget"
)
def has_model_budget() -> BudgetRow | None:
row = next(
(r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id),
None,
)
if row is None or not row.model_max_budget:
return None
return row if model_name in row.model_max_budget else None
_ = _poll(
client,
has_model_budget,
f"/budget/info never reported a model_max_budget entry for {model_name!r} "
f"on budget {budget_id}",
)
@pytest.mark.covers("mgmt.budget.update.persists")
def test_update_max_budget_persists_to_budget_info(
self, client: ManagementClient, resources: ResourceManager