From 1eaf4109894ede5c14858cc224933bea27cf6870 Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Sat, 25 Jul 2026 14:01:41 -0700 Subject: [PATCH 1/4] test(e2e): add failing reproducers for two open gateway bugs Both tests assert the behavior a customer expects and both are red today. They are reproducers, not regressions: the product is wrong, not the tests. Native passthrough returns almost none of the operational headers the managed route does. A /gemini/ generateContent call comes back with three x-litellm-* headers and no x-ratelimit-* at all, against sixteen and four on /v1beta/models/{m}:generateContent for the same prompt, and critically it omits x-litellm-response-cost. Customers front provider-native traffic through this route and read those headers to reconcile spend and pace themselves, so native traffic is currently invisible to the tooling that covers every other route. /budget/update rejects any model_max_budget with a 500. The reported symptom was model ids containing dots, and that reproduces (prisma raises "Unexpected `-5.2[FloatValue]` Expected `:`" because the key is interpolated into a GraphQL query unquoted, so glm-5.2 lexes as an identifier followed by a float), but the plain name gpt4o fails too, on a separate "model_max_budget should be of any of the following types: Json" type mismatch at budget_management_endpoints.py:173. Omitting the field returns 200. The test drives both names so the failure says whether per-model budgets are broken outright or only for punctuated ids; today it stops on the plain name, which is the wider bug. --- tests/e2e/coverage_registry/mgmt.yaml | 1 + .../llm_translation/test_passthrough_e2e.py | 32 ++++++++++ .../test_budget_customer_user_org_e2e.py | 61 ++++++++++++++++++- 3 files changed, 92 insertions(+), 2 deletions(-) diff --git a/tests/e2e/coverage_registry/mgmt.yaml b/tests/e2e/coverage_registry/mgmt.yaml index 68c5ef6b31d..af0ba9d56e6 100644 --- a/tests/e2e/coverage_registry/mgmt.yaml +++ b/tests/e2e/coverage_registry/mgmt.yaml @@ -54,6 +54,7 @@ - {id: mgmt.access_group.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "model_access_group_management_endpoints.py:600", rationale: "Access group membership query"} - {id: mgmt.mcp_server.register.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "mcp_management_endpoints.py:880", rationale: "MCP server registration"} - {id: mgmt.mcp_server.approve.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "mcp_management_endpoints.py:1200", rationale: "Admin approval persists"} +- {id: mgmt.budget.update.accepts_model_max_budget, module: mgmt, tier: P1, surface: api, assertions: [accepts_model_max_budget], source: "budget_management_endpoints.py:173", fail_before_fix: proven, rationale: "Per-model caps must be settable on an existing budget; model ids routinely carry dots and hyphens and the route must accept both"} - {id: mgmt.budget.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:155", rationale: "Limit changes apply"} - {id: mgmt.budget.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:280", rationale: "Clears limits"} - {id: mgmt.budget.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "budget_management_endpoints.py:215", rationale: "Budget enumeration"} diff --git a/tests/e2e/llm_translation/test_passthrough_e2e.py b/tests/e2e/llm_translation/test_passthrough_e2e.py index ed5c657d23e..016f3b7be41 100644 --- a/tests/e2e/llm_translation/test_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_e2e.py @@ -66,6 +66,38 @@ def test_gemini_passthrough_nonstreaming_logs_cost( assert tag in (row.request_tags or []), f"tags not logged: {row.request_tags}" +def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route( + client: PassthroughClient, scoped_key: str +) -> None: + """A native passthrough call must still be costable and pace-able by the client. + + Customers front provider-native traffic through /gemini/ and read the same + operational headers they get on /chat/completions: the response cost, so the + call reconciles against spend, and the x-ratelimit-* pacing headers, so a + client knows how much budget it has left. The passthrough route currently + returns neither, which makes native traffic invisible to the same tooling. + """ + result = client.gemini_generate( + scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}" + ) + require_successful_call(result) + + assert result.call_id, "passthrough must stamp x-litellm-call-id" + assert result.response_cost is not None, ( + "passthrough generateContent returned no x-litellm-response-cost header, so a " + "native call cannot be reconciled against spend the way /chat/completions can" + ) + assert result.response_cost > 0, ( + f"x-litellm-response-cost must be a real cost, got {result.response_cost}" + ) + + pacing = tuple(name for name in result.headers if name.startswith("x-ratelimit-")) + assert pacing, ( + "passthrough generateContent returned no x-ratelimit-* headers, so a client " + f"cannot pace itself; headers present were {sorted(result.headers)}" + ) + + def test_gemini_passthrough_streaming_logs_cost( client: PassthroughClient, scoped_key: str ) -> None: diff --git a/tests/e2e/management/test_budget_customer_user_org_e2e.py b/tests/e2e/management/test_budget_customer_user_org_e2e.py index 54cc18b228b..2fb52ecbe66 100644 --- a/tests/e2e/management/test_budget_customer_user_org_e2e.py +++ b/tests/e2e/management/test_budget_customer_user_org_e2e.py @@ -22,7 +22,7 @@ import pytest from pydantic import BaseModel, RootModel from e2e_config import unique_marker -from e2e_http import NoBody, unwrap +from e2e_http import NoBody, is_ok, unwrap from lifecycle import ResourceManager from management_client import ManagementClient from models import KeyGenerateBody, OrgInfoParams, OrgNewBody, UserNewBody @@ -53,9 +53,15 @@ class BudgetNewResponse(BaseModel): budget_id: str +class ModelBudgetEntry(BaseModel): + budget_limit: float + time_period: str + + class BudgetUpdateBody(BaseModel): budget_id: str - max_budget: float + max_budget: float | None = None + model_max_budget: dict[str, ModelBudgetEntry] | None = None class BudgetInfoBody(BaseModel): @@ -66,6 +72,7 @@ class BudgetRow(BaseModel): budget_id: str | None = None max_budget: float | None = None soft_budget: float | None = None + model_max_budget: dict[str, ModelBudgetEntry] | None = None class BudgetInfoResponse(RootModel[list[BudgetRow]]): @@ -148,6 +155,56 @@ class TestBudgetManagement: f"/budget/list never included the created budget {budget_id}", ) + @pytest.mark.covers("mgmt.budget.update.accepts_model_max_budget") + def test_update_accepts_per_model_budgets_including_punctuated_names( + self, client: ManagementClient, resources: ResourceManager + ) -> None: + """Per-model caps must be settable on an existing budget. + + `model_max_budget` is how a customer caps spend per model on a shared + budget, and model ids routinely carry dots and hyphens (`glm-5.2`). The + route has to accept both, and the plain name is included so a failure + says whether per-model budgets are broken outright or only for + punctuated ids. + """ + for model_name in ("gpt4o", "glm-5.2"): + budget_id = _create_budget( + client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET) + ) + + result = client.proxy.transport.post( + "/budget/update", + headers=client.proxy.transport.master, + json=BudgetUpdateBody( + budget_id=budget_id, + model_max_budget={ + model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d") + }, + ), + response_type=NoBody, + ) + + assert is_ok(result), ( + f"/budget/update rejected a per-model budget for {model_name!r}: {result}; " + f"a customer cannot cap spend per model on an existing budget" + ) + + def has_model_budget() -> BudgetRow | None: + row = next( + (r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), + None, + ) + if row is None or not row.model_max_budget: + return None + return row if model_name in row.model_max_budget else None + + _ = _poll( + client, + has_model_budget, + f"/budget/info never reported a model_max_budget entry for {model_name!r} " + f"on budget {budget_id}", + ) + @pytest.mark.covers("mgmt.budget.update.persists") def test_update_max_budget_persists_to_budget_info( self, client: ManagementClient, resources: ResourceManager From 78a759d73c6b11340a540d2ffa585b821670c927 Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Sat, 25 Jul 2026 14:10:03 -0700 Subject: [PATCH 2/4] test(e2e): add reproducer for unenforced end-user per-model rate limits model_max_budget accepts an rpm_limit alongside the spend cap, and /budget/new stores it: the create response echoes {"gemini-2.5-flash": {"rpm_limit": 1, "max_budget": 100.0, "budget_duration": "1d"}}. Attach that budget to an end user, drive three calls as that user, and all three return 200. The limit is accepted, persisted, and then ignored. The same shape already works when the budget hangs off a key, which is what makes this quietly dangerous: the API gives every indication the cap is in force. A customer using it to hold one end user to a slow rate on a shared key gets no throttling at all. Harness additions this needs: ModelBudgetEntry carries the rpm_limit/tpm_limit the route already accepts, BudgetNewBody and create_budget carry model_max_budget, and create_customer can attach an existing budget_id rather than only an inline max_budget. Red today, for the reason in the assertion message. --- .../coverage_registry/quota_management.yaml | 1 + tests/e2e/models.py | 2 + .../quota_management/budgets/budget_client.py | 22 +++++++-- .../budgets/test_model_max_budget_e2e.py | 47 +++++++++++++++++++ 4 files changed, 67 insertions(+), 5 deletions(-) diff --git a/tests/e2e/coverage_registry/quota_management.yaml b/tests/e2e/coverage_registry/quota_management.yaml index a8d0749cd8d..e853e631cee 100644 --- a/tests/e2e/coverage_registry/quota_management.yaml +++ b/tests/e2e/coverage_registry/quota_management.yaml @@ -18,6 +18,7 @@ - {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"} - {id: quota_management.budget.team_member.isolates_per_member, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [isolates_per_member], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "One team member's exhausted per-team budget does not block a different member on the same team"} - {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"} +- {id: quota_management.budget.end_user_model_max.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user_model_max, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "budget_management_endpoints.py", fail_before_fix: proven, rationale: "A per-model rpm_limit on an end-user budget is accepted and stored but never enforced; only key-attached budgets honour it"} - {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"} - {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"} - {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"} diff --git a/tests/e2e/models.py b/tests/e2e/models.py index af695acaa5e..a4e90c27f50 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -18,6 +18,8 @@ from pydantic import BaseModel, ConfigDict, RootModel, model_validator class ModelBudgetEntry(BaseModel): budget_limit: float time_period: str + rpm_limit: int | None = None + tpm_limit: int | None = None class BudgetWindow(BaseModel): diff --git a/tests/e2e/quota_management/budgets/budget_client.py b/tests/e2e/quota_management/budgets/budget_client.py index 83e8f27b597..5b9253928af 100644 --- a/tests/e2e/quota_management/budgets/budget_client.py +++ b/tests/e2e/quota_management/budgets/budget_client.py @@ -61,7 +61,8 @@ class UserDeleteBody(BaseModel): class CustomerNewBody(BaseModel): user_id: str - max_budget: float + max_budget: float | None = None + budget_id: str | None = None class OrgNewBody(BaseModel): @@ -151,9 +152,10 @@ class TagDeleteBody(BaseModel): class BudgetNewBody(BaseModel): - max_budget: float + max_budget: float | None = None soft_budget: float | None = None budget_duration: str | None = None + model_max_budget: dict[str, ModelBudgetEntry] | None = None class BudgetNewResponse(BaseModel): @@ -326,11 +328,19 @@ class BudgetClient: # ---- customer / end-user ------------------------------------------- - def create_customer(self, customer_id: str, *, max_budget: float) -> str: + def create_customer( + self, + customer_id: str, + *, + max_budget: float | None = None, + budget_id: str | None = None, + ) -> str: resp = self.proxy.transport.send( "/customer/new", headers=self.proxy.transport.master, - json=CustomerNewBody(user_id=customer_id, max_budget=max_budget), + json=CustomerNewBody( + user_id=customer_id, max_budget=max_budget, budget_id=budget_id + ), ) assert resp.ok, resp.body return customer_id @@ -509,9 +519,10 @@ class BudgetClient: def create_budget( self, *, - max_budget: float, + max_budget: float | None = None, soft_budget: float | None = None, budget_duration: str | None = None, + model_max_budget: dict[str, ModelBudgetEntry] | None = None, ) -> str: return unwrap( self.proxy.transport.post( @@ -521,6 +532,7 @@ class BudgetClient: max_budget=max_budget, soft_budget=soft_budget, budget_duration=budget_duration, + model_max_budget=model_max_budget, ), response_type=BudgetNewResponse, ) diff --git a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py index 4d0df2c35ea..eb08da62d6b 100644 --- a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py @@ -11,6 +11,7 @@ import time import pytest from budget_client import BudgetClient, is_budget_block, model_budget +from models import ModelBudgetEntry from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager @@ -56,3 +57,49 @@ def test_model_max_budget_isolates_per_model( f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated" ) require_successful_call(other) + + +@pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit") +def test_end_user_model_max_budget_enforces_per_model_rpm( + client: BudgetClient, resources: ResourceManager +) -> None: + """A per-model rpm_limit on an end user's budget has to actually throttle. + + `model_max_budget` accepts an `rpm_limit` alongside the spend cap, and a + customer uses it to hold one end user to a slow rate on an expensive model + without limiting the shared key everyone else runs through. The budget is + attached to the end user rather than to the key, which is the case that + matters here: the same shape already works when the budget hangs off a key. + """ + budget_id = client.create_budget( + model_max_budget={ + FREE_MODEL: ModelBudgetEntry( + budget_limit=1000.0, time_period="1d", rpm_limit=1 + ) + } + ) + resources.defer(lambda: client.delete_budget(budget_id)) + + customer = f"e2e-mmb-cust-{unique_marker()}" + _ = client.create_customer(customer, budget_id=budget_id) + resources.defer(lambda: client.delete_customers([customer])) + + key = client.generate_key() + resources.defer(lambda: client.delete_key(key)) + + statuses = tuple( + client.chat( + key, FREE_MODEL, f"hi {unique_marker()}", max_tokens=8, user=customer + ).status_code + for _ in range(3) + ) + + assert statuses[0] == 200, ( + f"the first call under an rpm_limit of 1 should succeed, got {statuses[0]}" + ) + assert 429 in statuses[1:], ( + f"an end-user budget with model_max_budget rpm_limit=1 did not throttle: " + f"three calls returned {statuses}. The limit is accepted and stored by " + f"/budget/new but never enforced for end-user budgets, so a customer " + f"cannot rate-limit an individual end user on a shared key" + ) From 340c43a2427f619c880f2df88c9e132551c6fdcf Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Mon, 10 Aug 2026 15:12:31 -0700 Subject: [PATCH 3/4] test(e2e): tighten model_max_budget reproducers and drop in-loop closure Trim the reproducer docstrings to the contract they assert, keeping the failure messages that document each red-by-design bug. Replace the nested per-model closure in the /budget/update test with a module-level predicate and a per-model helper so nothing closes over a loop variable, and fix the import order the merge left unsorted. --- .../llm_translation/test_passthrough_e2e.py | 11 +-- .../test_budget_customer_user_org_e2e.py | 81 ++++++++++--------- .../budgets/test_model_max_budget_e2e.py | 13 ++- 3 files changed, 54 insertions(+), 51 deletions(-) diff --git a/tests/e2e/llm_translation/test_passthrough_e2e.py b/tests/e2e/llm_translation/test_passthrough_e2e.py index 016f3b7be41..081e88e493b 100644 --- a/tests/e2e/llm_translation/test_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_e2e.py @@ -69,13 +69,10 @@ def test_gemini_passthrough_nonstreaming_logs_cost( def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route( client: PassthroughClient, scoped_key: str ) -> None: - """A native passthrough call must still be costable and pace-able by the client. - - Customers front provider-native traffic through /gemini/ and read the same - operational headers they get on /chat/completions: the response cost, so the - call reconciles against spend, and the x-ratelimit-* pacing headers, so a - client knows how much budget it has left. The passthrough route currently - returns neither, which makes native traffic invisible to the same tooling. + """Native /gemini/ passthrough must return the same operational headers as + /chat/completions: x-litellm-response-cost so the call reconciles against + spend, and x-ratelimit-* so a client can pace itself. It returns neither + today, which makes native traffic invisible to the same tooling. """ result = client.gemini_generate( scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}" diff --git a/tests/e2e/management/test_budget_customer_user_org_e2e.py b/tests/e2e/management/test_budget_customer_user_org_e2e.py index 607c7094654..cc5423affeb 100644 --- a/tests/e2e/management/test_budget_customer_user_org_e2e.py +++ b/tests/e2e/management/test_budget_customer_user_org_e2e.py @@ -125,6 +125,17 @@ def _budget_rows(client: ManagementClient, budget_id: str) -> tuple[BudgetRow, . ) +def _find_model_budget( + client: ManagementClient, budget_id: str, model_name: str +) -> BudgetRow | None: + row = next( + (r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), None + ) + if row is None or not row.model_max_budget or model_name not in row.model_max_budget: + return None + return row + + def _budget_list_ids(client: ManagementClient) -> tuple[str, ...]: return tuple( row.budget_id @@ -161,51 +172,47 @@ class TestBudgetManagement: def test_update_accepts_per_model_budgets_including_punctuated_names( self, client: ManagementClient, resources: ResourceManager ) -> None: - """Per-model caps must be settable on an existing budget. + """/budget/update must accept per-model caps on an existing budget. - `model_max_budget` is how a customer caps spend per model on a shared - budget, and model ids routinely carry dots and hyphens (`glm-5.2`). The - route has to accept both, and the plain name is included so a failure - says whether per-model budgets are broken outright or only for + model_max_budget keys are model ids, which routinely carry dots and + hyphens (glm-5.2). Both a plain and a punctuated id are exercised so a + failure says whether per-model budgets break outright or only for punctuated ids. """ for model_name in ("gpt4o", "glm-5.2"): - budget_id = _create_budget( - client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET) - ) + self._assert_model_budget_round_trips(client, resources, model_name) - result = client.proxy.transport.post( - "/budget/update", - headers=client.proxy.transport.master, - json=BudgetUpdateBody( - budget_id=budget_id, - model_max_budget={ - model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d") - }, - ), - response_type=NoBody, - ) + @staticmethod + def _assert_model_budget_round_trips( + client: ManagementClient, resources: ResourceManager, model_name: str + ) -> None: + budget_id = _create_budget( + client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET) + ) - assert is_ok(result), ( - f"/budget/update rejected a per-model budget for {model_name!r}: {result}; " - f"a customer cannot cap spend per model on an existing budget" - ) + result = client.proxy.transport.post( + "/budget/update", + headers=client.proxy.transport.master, + json=BudgetUpdateBody( + budget_id=budget_id, + model_max_budget={ + model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d") + }, + ), + response_type=NoBody, + ) - def has_model_budget() -> BudgetRow | None: - row = next( - (r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), - None, - ) - if row is None or not row.model_max_budget: - return None - return row if model_name in row.model_max_budget else None + assert is_ok(result), ( + f"/budget/update rejected a per-model budget for {model_name!r}: {result}; " + f"a customer cannot cap spend per model on an existing budget" + ) - _ = _poll( - client, - has_model_budget, - f"/budget/info never reported a model_max_budget entry for {model_name!r} " - f"on budget {budget_id}", - ) + _ = _poll( + client, + lambda: _find_model_budget(client, budget_id, model_name), + f"/budget/info never reported a model_max_budget entry for {model_name!r} " + f"on budget {budget_id}", + ) @pytest.mark.covers("mgmt.budget.update.persists") def test_update_max_budget_persists_to_budget_info( diff --git a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py index eb08da62d6b..f202bb570b6 100644 --- a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py @@ -11,10 +11,10 @@ import time import pytest from budget_client import BudgetClient, is_budget_block, model_budget -from models import ModelBudgetEntry from e2e_config import unique_marker from e2e_http import require_successful_call from lifecycle import ResourceManager +from models import ModelBudgetEntry pytestmark = pytest.mark.e2e @@ -63,13 +63,12 @@ def test_model_max_budget_isolates_per_model( def test_end_user_model_max_budget_enforces_per_model_rpm( client: BudgetClient, resources: ResourceManager ) -> None: - """A per-model rpm_limit on an end user's budget has to actually throttle. + """A per-model rpm_limit on an end-user budget must actually throttle. - `model_max_budget` accepts an `rpm_limit` alongside the spend cap, and a - customer uses it to hold one end user to a slow rate on an expensive model - without limiting the shared key everyone else runs through. The budget is - attached to the end user rather than to the key, which is the case that - matters here: the same shape already works when the budget hangs off a key. + model_max_budget takes an rpm_limit alongside the spend cap, letting a + customer hold one end user to a slow rate without limiting the shared key. + The budget hangs off the end user, not the key; the key-attached shape + already works, so this pins the end-user gap. """ budget_id = client.create_budget( model_max_budget={ From 85cf3d77c38a77a9e8ec464ef4f8bdc5f7932a7b Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Mon, 10 Aug 2026 22:57:22 -0700 Subject: [PATCH 4/4] test(e2e): skip the three reproducers while their gateway bugs stay open The passthrough header contract, /budget/update model_max_budget, and end-user per-model rpm enforcement reproducers all still fail against staging by design. Skip each with the product gap named so the combined suite can gate merges on green while the collector keeps reporting the cells as uncovered. --- tests/e2e/llm_translation/test_passthrough_e2e.py | 1 + tests/e2e/management/test_budget_customer_user_org_e2e.py | 1 + tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py | 1 + 3 files changed, 3 insertions(+) diff --git a/tests/e2e/llm_translation/test_passthrough_e2e.py b/tests/e2e/llm_translation/test_passthrough_e2e.py index 081e88e493b..b57164df9bb 100644 --- a/tests/e2e/llm_translation/test_passthrough_e2e.py +++ b/tests/e2e/llm_translation/test_passthrough_e2e.py @@ -66,6 +66,7 @@ def test_gemini_passthrough_nonstreaming_logs_cost( assert tag in (row.request_tags or []), f"tags not logged: {row.request_tags}" +@pytest.mark.skip(reason="stage red: product gap, native passthrough returns no x-litellm-response-cost or x-ratelimit-* headers") def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route( client: PassthroughClient, scoped_key: str ) -> None: diff --git a/tests/e2e/management/test_budget_customer_user_org_e2e.py b/tests/e2e/management/test_budget_customer_user_org_e2e.py index cc5423affeb..ae78555de78 100644 --- a/tests/e2e/management/test_budget_customer_user_org_e2e.py +++ b/tests/e2e/management/test_budget_customer_user_org_e2e.py @@ -168,6 +168,7 @@ class TestBudgetManagement: f"/budget/list never included the created budget {budget_id}", ) + @pytest.mark.skip(reason="stage red: product gap, /budget/update 500s on any model_max_budget (prisma Json arg + unquoted GraphQL interpolation)") @pytest.mark.covers("mgmt.budget.update.accepts_model_max_budget") def test_update_accepts_per_model_budgets_including_punctuated_names( self, client: ManagementClient, resources: ResourceManager diff --git a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py index f202bb570b6..3f7264365ef 100644 --- a/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py +++ b/tests/e2e/quota_management/budgets/test_model_max_budget_e2e.py @@ -59,6 +59,7 @@ def test_model_max_budget_isolates_per_model( require_successful_call(other) +@pytest.mark.skip(reason="stage red: product gap, end-user model_max_budget rpm_limit is stored but never enforced") @pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit") def test_end_user_model_max_budget_enforces_per_model_rpm( client: BudgetClient, resources: ResourceManager