mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
test(e2e): tighten model_max_budget reproducers and drop in-loop closure
Trim the reproducer docstrings to the contract they assert, keeping the failure messages that document each red-by-design bug. Replace the nested per-model closure in the /budget/update test with a module-level predicate and a per-model helper so nothing closes over a loop variable, and fix the import order the merge left unsorted.
This commit is contained in:
parent
aa7b0b3cd2
commit
340c43a242
3 changed files with 54 additions and 51 deletions
|
|
@ -69,13 +69,10 @@ def test_gemini_passthrough_nonstreaming_logs_cost(
|
|||
def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
"""A native passthrough call must still be costable and pace-able by the client.
|
||||
|
||||
Customers front provider-native traffic through /gemini/ and read the same
|
||||
operational headers they get on /chat/completions: the response cost, so the
|
||||
call reconciles against spend, and the x-ratelimit-* pacing headers, so a
|
||||
client knows how much budget it has left. The passthrough route currently
|
||||
returns neither, which makes native traffic invisible to the same tooling.
|
||||
"""Native /gemini/ passthrough must return the same operational headers as
|
||||
/chat/completions: x-litellm-response-cost so the call reconciles against
|
||||
spend, and x-ratelimit-* so a client can pace itself. It returns neither
|
||||
today, which makes native traffic invisible to the same tooling.
|
||||
"""
|
||||
result = client.gemini_generate(
|
||||
scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}"
|
||||
|
|
|
|||
|
|
@ -125,6 +125,17 @@ def _budget_rows(client: ManagementClient, budget_id: str) -> tuple[BudgetRow, .
|
|||
)
|
||||
|
||||
|
||||
def _find_model_budget(
|
||||
client: ManagementClient, budget_id: str, model_name: str
|
||||
) -> BudgetRow | None:
|
||||
row = next(
|
||||
(r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), None
|
||||
)
|
||||
if row is None or not row.model_max_budget or model_name not in row.model_max_budget:
|
||||
return None
|
||||
return row
|
||||
|
||||
|
||||
def _budget_list_ids(client: ManagementClient) -> tuple[str, ...]:
|
||||
return tuple(
|
||||
row.budget_id
|
||||
|
|
@ -161,51 +172,47 @@ class TestBudgetManagement:
|
|||
def test_update_accepts_per_model_budgets_including_punctuated_names(
|
||||
self, client: ManagementClient, resources: ResourceManager
|
||||
) -> None:
|
||||
"""Per-model caps must be settable on an existing budget.
|
||||
"""/budget/update must accept per-model caps on an existing budget.
|
||||
|
||||
`model_max_budget` is how a customer caps spend per model on a shared
|
||||
budget, and model ids routinely carry dots and hyphens (`glm-5.2`). The
|
||||
route has to accept both, and the plain name is included so a failure
|
||||
says whether per-model budgets are broken outright or only for
|
||||
model_max_budget keys are model ids, which routinely carry dots and
|
||||
hyphens (glm-5.2). Both a plain and a punctuated id are exercised so a
|
||||
failure says whether per-model budgets break outright or only for
|
||||
punctuated ids.
|
||||
"""
|
||||
for model_name in ("gpt4o", "glm-5.2"):
|
||||
budget_id = _create_budget(
|
||||
client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET)
|
||||
)
|
||||
self._assert_model_budget_round_trips(client, resources, model_name)
|
||||
|
||||
result = client.proxy.transport.post(
|
||||
"/budget/update",
|
||||
headers=client.proxy.transport.master,
|
||||
json=BudgetUpdateBody(
|
||||
budget_id=budget_id,
|
||||
model_max_budget={
|
||||
model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d")
|
||||
},
|
||||
),
|
||||
response_type=NoBody,
|
||||
)
|
||||
@staticmethod
|
||||
def _assert_model_budget_round_trips(
|
||||
client: ManagementClient, resources: ResourceManager, model_name: str
|
||||
) -> None:
|
||||
budget_id = _create_budget(
|
||||
client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET)
|
||||
)
|
||||
|
||||
assert is_ok(result), (
|
||||
f"/budget/update rejected a per-model budget for {model_name!r}: {result}; "
|
||||
f"a customer cannot cap spend per model on an existing budget"
|
||||
)
|
||||
result = client.proxy.transport.post(
|
||||
"/budget/update",
|
||||
headers=client.proxy.transport.master,
|
||||
json=BudgetUpdateBody(
|
||||
budget_id=budget_id,
|
||||
model_max_budget={
|
||||
model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d")
|
||||
},
|
||||
),
|
||||
response_type=NoBody,
|
||||
)
|
||||
|
||||
def has_model_budget() -> BudgetRow | None:
|
||||
row = next(
|
||||
(r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id),
|
||||
None,
|
||||
)
|
||||
if row is None or not row.model_max_budget:
|
||||
return None
|
||||
return row if model_name in row.model_max_budget else None
|
||||
assert is_ok(result), (
|
||||
f"/budget/update rejected a per-model budget for {model_name!r}: {result}; "
|
||||
f"a customer cannot cap spend per model on an existing budget"
|
||||
)
|
||||
|
||||
_ = _poll(
|
||||
client,
|
||||
has_model_budget,
|
||||
f"/budget/info never reported a model_max_budget entry for {model_name!r} "
|
||||
f"on budget {budget_id}",
|
||||
)
|
||||
_ = _poll(
|
||||
client,
|
||||
lambda: _find_model_budget(client, budget_id, model_name),
|
||||
f"/budget/info never reported a model_max_budget entry for {model_name!r} "
|
||||
f"on budget {budget_id}",
|
||||
)
|
||||
|
||||
@pytest.mark.covers("mgmt.budget.update.persists")
|
||||
def test_update_max_budget_persists_to_budget_info(
|
||||
|
|
|
|||
|
|
@ -11,10 +11,10 @@ import time
|
|||
import pytest
|
||||
|
||||
from budget_client import BudgetClient, is_budget_block, model_budget
|
||||
from models import ModelBudgetEntry
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import require_successful_call
|
||||
from lifecycle import ResourceManager
|
||||
from models import ModelBudgetEntry
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
|
@ -63,13 +63,12 @@ def test_model_max_budget_isolates_per_model(
|
|||
def test_end_user_model_max_budget_enforces_per_model_rpm(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
"""A per-model rpm_limit on an end user's budget has to actually throttle.
|
||||
"""A per-model rpm_limit on an end-user budget must actually throttle.
|
||||
|
||||
`model_max_budget` accepts an `rpm_limit` alongside the spend cap, and a
|
||||
customer uses it to hold one end user to a slow rate on an expensive model
|
||||
without limiting the shared key everyone else runs through. The budget is
|
||||
attached to the end user rather than to the key, which is the case that
|
||||
matters here: the same shape already works when the budget hangs off a key.
|
||||
model_max_budget takes an rpm_limit alongside the spend cap, letting a
|
||||
customer hold one end user to a slow rate without limiting the shared key.
|
||||
The budget hangs off the end user, not the key; the key-attached shape
|
||||
already works, so this pins the end-user gap.
|
||||
"""
|
||||
budget_id = client.create_budget(
|
||||
model_max_budget={
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue