mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
merge(e2e): PR #34657 into combined e2e run branch
This commit is contained in:
commit
d593ddb26e
7 changed files with 165 additions and 7 deletions
|
|
@ -54,6 +54,7 @@
|
|||
- {id: mgmt.access_group.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "model_access_group_management_endpoints.py:600", rationale: "Access group membership query"}
|
||||
- {id: mgmt.mcp_server.register.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "mcp_management_endpoints.py:880", rationale: "MCP server registration"}
|
||||
- {id: mgmt.mcp_server.approve.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "mcp_management_endpoints.py:1200", rationale: "Admin approval persists"}
|
||||
- {id: mgmt.budget.update.accepts_model_max_budget, module: mgmt, tier: P1, surface: api, assertions: [accepts_model_max_budget], source: "budget_management_endpoints.py:173", fail_before_fix: proven, rationale: "Per-model caps must be settable on an existing budget; model ids routinely carry dots and hyphens and the route must accept both"}
|
||||
- {id: mgmt.budget.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:155", rationale: "Limit changes apply"}
|
||||
- {id: mgmt.budget.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "budget_management_endpoints.py:280", rationale: "Clears limits"}
|
||||
- {id: mgmt.budget.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "budget_management_endpoints.py:215", rationale: "Budget enumeration"}
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@
|
|||
- {id: quota_management.budget.team_member.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "A member's per-team budget blocks independently of the team budget"}
|
||||
- {id: quota_management.budget.team_member.isolates_per_member, module: quota_management, tier: P1, behavior: budget, variant: team_member, assertions: [isolates_per_member], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "One team member's exhausted per-team budget does not block a different member on the same team"}
|
||||
- {id: quota_management.budget.tag.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: tag, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "router_strategy/budget_limiter.py", rationale: "Proxy-level tag budgets block tagged requests at the cap"}
|
||||
- {id: quota_management.budget.end_user_model_max.blocks_over_limit, module: quota_management, tier: P1, behavior: budget, variant: end_user_model_max, assertions: [blocks_over_limit], exercised_on: [chat_completions], source: "budget_management_endpoints.py", fail_before_fix: proven, rationale: "A per-model rpm_limit on an end-user budget is accepted and stored but never enforced; only key-attached budgets honour it"}
|
||||
- {id: quota_management.budget.model_max.isolates_per_model, module: quota_management, tier: P1, behavior: budget, variant: model_max, assertions: [isolates_per_model], exercised_on: [chat_completions], source: "proxy/hooks/model_max_budget_limiter.py", rationale: "model_max_budget caps one model without touching a sibling's budget"}
|
||||
- {id: quota_management.budget.soft.alerts_without_blocking, module: quota_management, tier: P1, behavior: budget, variant: soft, assertions: [alerts_without_blocking], exercised_on: [chat_completions], source: "proxy/auth/auth_checks.py", rationale: "soft_budget alerts but never blocks traffic"}
|
||||
- {id: quota_management.budget.key.resets_after_window, module: quota_management, tier: P1, behavior: budget, variant: key, assertions: [resets_after_window], exercised_on: [chat_completions], source: "proxy/common_utils/reset_budget_job.py", rationale: "budget_duration zeroes key spend after the window; a blocked key serves again"}
|
||||
|
|
|
|||
|
|
@ -66,6 +66,36 @@ def test_gemini_passthrough_nonstreaming_logs_cost(
|
|||
assert tag in (row.request_tags or []), f"tags not logged: {row.request_tags}"
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="stage red: product gap, native passthrough returns no x-litellm-response-cost or x-ratelimit-* headers")
|
||||
def test_gemini_passthrough_returns_the_same_header_contract_as_the_managed_route(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
"""Native /gemini/ passthrough must return the same operational headers as
|
||||
/chat/completions: x-litellm-response-cost so the call reconciles against
|
||||
spend, and x-ratelimit-* so a client can pace itself. It returns neither
|
||||
today, which makes native traffic invisible to the same tooling.
|
||||
"""
|
||||
result = client.gemini_generate(
|
||||
scoped_key, "gemini-2.5-flash", f"Say hello in one word. {unique_marker()}"
|
||||
)
|
||||
require_successful_call(result)
|
||||
|
||||
assert result.call_id, "passthrough must stamp x-litellm-call-id"
|
||||
assert result.response_cost is not None, (
|
||||
"passthrough generateContent returned no x-litellm-response-cost header, so a "
|
||||
"native call cannot be reconciled against spend the way /chat/completions can"
|
||||
)
|
||||
assert result.response_cost > 0, (
|
||||
f"x-litellm-response-cost must be a real cost, got {result.response_cost}"
|
||||
)
|
||||
|
||||
pacing = tuple(name for name in result.headers if name.startswith("x-ratelimit-"))
|
||||
assert pacing, (
|
||||
"passthrough generateContent returned no x-ratelimit-* headers, so a client "
|
||||
f"cannot pace itself; headers present were {sorted(result.headers)}"
|
||||
)
|
||||
|
||||
|
||||
def test_gemini_passthrough_streaming_logs_cost(
|
||||
client: PassthroughClient, scoped_key: str
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ import pytest
|
|||
from pydantic import BaseModel, Field, RootModel
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import NoBody, Success, UnauthorizedError, UnknownApiError, unwrap
|
||||
from e2e_http import NoBody, Success, UnauthorizedError, UnknownApiError, is_ok, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from management_client import ManagementClient
|
||||
from models import KeyGenerateBody, OrgInfoParams, OrgNewBody, UserNewBody
|
||||
|
|
@ -55,9 +55,15 @@ class BudgetNewResponse(BaseModel):
|
|||
budget_id: str
|
||||
|
||||
|
||||
class ModelBudgetEntry(BaseModel):
|
||||
budget_limit: float
|
||||
time_period: str
|
||||
|
||||
|
||||
class BudgetUpdateBody(BaseModel):
|
||||
budget_id: str
|
||||
max_budget: float
|
||||
max_budget: float | None = None
|
||||
model_max_budget: dict[str, ModelBudgetEntry] | None = None
|
||||
|
||||
|
||||
class BudgetInfoBody(BaseModel):
|
||||
|
|
@ -68,6 +74,7 @@ class BudgetRow(BaseModel):
|
|||
budget_id: str | None = None
|
||||
max_budget: float | None = None
|
||||
soft_budget: float | None = None
|
||||
model_max_budget: dict[str, ModelBudgetEntry] | None = None
|
||||
|
||||
|
||||
class BudgetInfoResponse(RootModel[list[BudgetRow]]):
|
||||
|
|
@ -118,6 +125,17 @@ def _budget_rows(client: ManagementClient, budget_id: str) -> tuple[BudgetRow, .
|
|||
)
|
||||
|
||||
|
||||
def _find_model_budget(
|
||||
client: ManagementClient, budget_id: str, model_name: str
|
||||
) -> BudgetRow | None:
|
||||
row = next(
|
||||
(r for r in _budget_rows(client, budget_id) if r.budget_id == budget_id), None
|
||||
)
|
||||
if row is None or not row.model_max_budget or model_name not in row.model_max_budget:
|
||||
return None
|
||||
return row
|
||||
|
||||
|
||||
def _budget_list_ids(client: ManagementClient) -> tuple[str, ...]:
|
||||
return tuple(
|
||||
row.budget_id
|
||||
|
|
@ -150,6 +168,53 @@ class TestBudgetManagement:
|
|||
f"/budget/list never included the created budget {budget_id}",
|
||||
)
|
||||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /budget/update 500s on any model_max_budget (prisma Json arg + unquoted GraphQL interpolation)")
|
||||
@pytest.mark.covers("mgmt.budget.update.accepts_model_max_budget")
|
||||
def test_update_accepts_per_model_budgets_including_punctuated_names(
|
||||
self, client: ManagementClient, resources: ResourceManager
|
||||
) -> None:
|
||||
"""/budget/update must accept per-model caps on an existing budget.
|
||||
|
||||
model_max_budget keys are model ids, which routinely carry dots and
|
||||
hyphens (glm-5.2). Both a plain and a punctuated id are exercised so a
|
||||
failure says whether per-model budgets break outright or only for
|
||||
punctuated ids.
|
||||
"""
|
||||
for model_name in ("gpt4o", "glm-5.2"):
|
||||
self._assert_model_budget_round_trips(client, resources, model_name)
|
||||
|
||||
@staticmethod
|
||||
def _assert_model_budget_round_trips(
|
||||
client: ManagementClient, resources: ResourceManager, model_name: str
|
||||
) -> None:
|
||||
budget_id = _create_budget(
|
||||
client, resources, BudgetNewBody(max_budget=_INITIAL_MAX_BUDGET)
|
||||
)
|
||||
|
||||
result = client.proxy.transport.post(
|
||||
"/budget/update",
|
||||
headers=client.proxy.transport.master,
|
||||
json=BudgetUpdateBody(
|
||||
budget_id=budget_id,
|
||||
model_max_budget={
|
||||
model_name: ModelBudgetEntry(budget_limit=5.0, time_period="1d")
|
||||
},
|
||||
),
|
||||
response_type=NoBody,
|
||||
)
|
||||
|
||||
assert is_ok(result), (
|
||||
f"/budget/update rejected a per-model budget for {model_name!r}: {result}; "
|
||||
f"a customer cannot cap spend per model on an existing budget"
|
||||
)
|
||||
|
||||
_ = _poll(
|
||||
client,
|
||||
lambda: _find_model_budget(client, budget_id, model_name),
|
||||
f"/budget/info never reported a model_max_budget entry for {model_name!r} "
|
||||
f"on budget {budget_id}",
|
||||
)
|
||||
|
||||
@pytest.mark.covers("mgmt.budget.update.persists")
|
||||
def test_update_max_budget_persists_to_budget_info(
|
||||
self, client: ManagementClient, resources: ResourceManager
|
||||
|
|
|
|||
|
|
@ -18,6 +18,8 @@ from pydantic import BaseModel, ConfigDict, RootModel, model_validator
|
|||
class ModelBudgetEntry(BaseModel):
|
||||
budget_limit: float
|
||||
time_period: str
|
||||
rpm_limit: int | None = None
|
||||
tpm_limit: int | None = None
|
||||
|
||||
|
||||
class BudgetWindow(BaseModel):
|
||||
|
|
|
|||
|
|
@ -61,7 +61,8 @@ class UserDeleteBody(BaseModel):
|
|||
|
||||
class CustomerNewBody(BaseModel):
|
||||
user_id: str
|
||||
max_budget: float
|
||||
max_budget: float | None = None
|
||||
budget_id: str | None = None
|
||||
|
||||
|
||||
class OrgNewBody(BaseModel):
|
||||
|
|
@ -151,9 +152,10 @@ class TagDeleteBody(BaseModel):
|
|||
|
||||
|
||||
class BudgetNewBody(BaseModel):
|
||||
max_budget: float
|
||||
max_budget: float | None = None
|
||||
soft_budget: float | None = None
|
||||
budget_duration: str | None = None
|
||||
model_max_budget: dict[str, ModelBudgetEntry] | None = None
|
||||
|
||||
|
||||
class BudgetNewResponse(BaseModel):
|
||||
|
|
@ -326,11 +328,19 @@ class BudgetClient:
|
|||
|
||||
# ---- customer / end-user -------------------------------------------
|
||||
|
||||
def create_customer(self, customer_id: str, *, max_budget: float) -> str:
|
||||
def create_customer(
|
||||
self,
|
||||
customer_id: str,
|
||||
*,
|
||||
max_budget: float | None = None,
|
||||
budget_id: str | None = None,
|
||||
) -> str:
|
||||
resp = self.proxy.transport.send(
|
||||
"/customer/new",
|
||||
headers=self.proxy.transport.master,
|
||||
json=CustomerNewBody(user_id=customer_id, max_budget=max_budget),
|
||||
json=CustomerNewBody(
|
||||
user_id=customer_id, max_budget=max_budget, budget_id=budget_id
|
||||
),
|
||||
)
|
||||
assert resp.ok, resp.body
|
||||
return customer_id
|
||||
|
|
@ -509,9 +519,10 @@ class BudgetClient:
|
|||
def create_budget(
|
||||
self,
|
||||
*,
|
||||
max_budget: float,
|
||||
max_budget: float | None = None,
|
||||
soft_budget: float | None = None,
|
||||
budget_duration: str | None = None,
|
||||
model_max_budget: dict[str, ModelBudgetEntry] | None = None,
|
||||
) -> str:
|
||||
return unwrap(
|
||||
self.proxy.transport.post(
|
||||
|
|
@ -521,6 +532,7 @@ class BudgetClient:
|
|||
max_budget=max_budget,
|
||||
soft_budget=soft_budget,
|
||||
budget_duration=budget_duration,
|
||||
model_max_budget=model_max_budget,
|
||||
),
|
||||
response_type=BudgetNewResponse,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ from budget_client import BudgetClient, is_budget_block, model_budget
|
|||
from e2e_config import unique_marker
|
||||
from e2e_http import require_successful_call
|
||||
from lifecycle import ResourceManager
|
||||
from models import ModelBudgetEntry
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
|
|
@ -56,3 +57,49 @@ def test_model_max_budget_isolates_per_model(
|
|||
f"{FREE_MODEL} was blocked by {CAPPED_MODEL}'s budget; per-model caps not isolated"
|
||||
)
|
||||
require_successful_call(other)
|
||||
|
||||
|
||||
@pytest.mark.skip(reason="stage red: product gap, end-user model_max_budget rpm_limit is stored but never enforced")
|
||||
@pytest.mark.covers("quota_management.budget.end_user_model_max.blocks_over_limit")
|
||||
def test_end_user_model_max_budget_enforces_per_model_rpm(
|
||||
client: BudgetClient, resources: ResourceManager
|
||||
) -> None:
|
||||
"""A per-model rpm_limit on an end-user budget must actually throttle.
|
||||
|
||||
model_max_budget takes an rpm_limit alongside the spend cap, letting a
|
||||
customer hold one end user to a slow rate without limiting the shared key.
|
||||
The budget hangs off the end user, not the key; the key-attached shape
|
||||
already works, so this pins the end-user gap.
|
||||
"""
|
||||
budget_id = client.create_budget(
|
||||
model_max_budget={
|
||||
FREE_MODEL: ModelBudgetEntry(
|
||||
budget_limit=1000.0, time_period="1d", rpm_limit=1
|
||||
)
|
||||
}
|
||||
)
|
||||
resources.defer(lambda: client.delete_budget(budget_id))
|
||||
|
||||
customer = f"e2e-mmb-cust-{unique_marker()}"
|
||||
_ = client.create_customer(customer, budget_id=budget_id)
|
||||
resources.defer(lambda: client.delete_customers([customer]))
|
||||
|
||||
key = client.generate_key()
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
statuses = tuple(
|
||||
client.chat(
|
||||
key, FREE_MODEL, f"hi {unique_marker()}", max_tokens=8, user=customer
|
||||
).status_code
|
||||
for _ in range(3)
|
||||
)
|
||||
|
||||
assert statuses[0] == 200, (
|
||||
f"the first call under an rpm_limit of 1 should succeed, got {statuses[0]}"
|
||||
)
|
||||
assert 429 in statuses[1:], (
|
||||
f"an end-user budget with model_max_budget rpm_limit=1 did not throttle: "
|
||||
f"three calls returned {statuses}. The limit is accepted and stored by "
|
||||
f"/budget/new but never enforced for end-user budgets, so a customer "
|
||||
f"cannot rate-limit an individual end user on a shared key"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue