mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(team): match provider-prefixed team budget entries and stabilize CI
Team model_max_budget entries keyed provider/model (e.g. openai/gpt-4) never matched bare model-group requests because only the request side was provider-normalized; the cap was silently unenforced. The matcher now strips the prefix on both sides and keeps the config entry name as the canonical spend counter key. DefaultInternalUserParams declared the same four-role Literal in a different member order than the three declarations at the top of _types.py; typing's Literal cache makes the surviving order Python-version dependent, so local and CI runs generated schema.d.ts with conflicting enum orders. All declarations now share one order. The script-mode proxy_cli regression test performed a real proxy_server import in a subprocess and timed out on CI runners; both module names are now pre-stubbed in sys.modules (matching the other run_server tests) so the good path stays fast and a regression to the bare import fails fast with ImportError.
This commit is contained in:
parent
2abdae078d
commit
5cc8c1b1a2
5 changed files with 91 additions and 13 deletions
|
|
@ -4522,10 +4522,10 @@ class DefaultInternalUserParams(LiteLLMPydanticObjectBase):
|
|||
|
||||
user_role: (
|
||||
Literal[
|
||||
LitellmUserRoles.INTERNAL_USER,
|
||||
LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
|
||||
LitellmUserRoles.PROXY_ADMIN,
|
||||
LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
|
||||
LitellmUserRoles.INTERNAL_USER,
|
||||
LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
|
||||
]
|
||||
| None
|
||||
) = Field(
|
||||
|
|
|
|||
|
|
@ -162,16 +162,24 @@ class _PROXY_VirtualKeyModelMaxBudgetLimiter(RouterBudgetLimiting):
|
|||
self, model: str, internal_model_max_budget: Mapping[str, BudgetConfig]
|
||||
) -> str | None:
|
||||
"""
|
||||
Resolve the budget-config entry name `model` matches (exact, then
|
||||
`{provider}/{model}`-normalized). The entry name is the canonical form for
|
||||
spend counter keys so prefixed and bare spellings of one model share a counter.
|
||||
Resolve the budget-config entry name `model` matches: exact first, then with
|
||||
the `{provider}/` prefix stripped from the request, the entry, or both. The
|
||||
entry name is the canonical form for spend counter keys so prefixed and bare
|
||||
spellings of one model share a counter.
|
||||
"""
|
||||
if model in internal_model_max_budget:
|
||||
return model
|
||||
stripped_model = self._get_model_without_custom_llm_provider(model)
|
||||
if stripped_model in internal_model_max_budget:
|
||||
return stripped_model
|
||||
return None
|
||||
return next(
|
||||
(
|
||||
entry_name
|
||||
for entry_name in internal_model_max_budget
|
||||
if self._get_model_without_custom_llm_provider(entry_name) in (model, stripped_model)
|
||||
),
|
||||
None,
|
||||
)
|
||||
|
||||
def _key_already_covers_model(
|
||||
self, key_model_max_budget: Mapping[str, Mapping[str, str | float]] | None, model: str
|
||||
|
|
|
|||
|
|
@ -966,3 +966,61 @@ async def test_team_window_start_key_is_per_model_and_duration(budget_limiter):
|
|||
"team_model_budget_start_time:team-windows:gpt-4:1d",
|
||||
"team_model_budget_start_time:team-windows:claude-3:7d",
|
||||
]
|
||||
|
||||
|
||||
def test_get_matched_budget_model_name_strips_entry_prefix(budget_limiter):
|
||||
"""Config entries keyed provider/model must bind bare (and differently prefixed)
|
||||
requests; before the fix only the request side was normalized, so an
|
||||
openai/gpt-4 entry never matched a gpt-4 request (Greptile finding)."""
|
||||
configs = budget_limiter._coerce_budget_configs(
|
||||
{"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}}
|
||||
)
|
||||
assert (
|
||||
budget_limiter._get_matched_budget_model_name(
|
||||
model="gpt-4", internal_model_max_budget=configs
|
||||
)
|
||||
== "openai/gpt-4"
|
||||
)
|
||||
assert (
|
||||
budget_limiter._get_matched_budget_model_name(
|
||||
model="azure/gpt-4", internal_model_max_budget=configs
|
||||
)
|
||||
== "openai/gpt-4"
|
||||
)
|
||||
assert (
|
||||
budget_limiter._get_matched_budget_model_name(
|
||||
model="claude-3", internal_model_max_budget=configs
|
||||
)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_prefixed_team_budget_entry_matches_bare_request(budget_limiter):
|
||||
"""A team cap keyed openai/gpt-4 must govern a bare gpt-4 request end to end:
|
||||
spend recorded under the canonical entry name, then enforcement raising for
|
||||
either spelling. Uses the real DualCache, no spend-read mocks."""
|
||||
team_budget = {"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}}
|
||||
kwargs = {
|
||||
"standard_logging_object": {
|
||||
"response_cost": 6.0,
|
||||
"model_group": "gpt-4",
|
||||
"model": "gpt-4",
|
||||
"metadata": {"user_api_key_hash": "hash-1"},
|
||||
},
|
||||
"litellm_params": {
|
||||
"metadata": {
|
||||
"user_api_key_team_id": "team-prefixed",
|
||||
"user_api_key_team_model_max_budget": team_budget,
|
||||
}
|
||||
},
|
||||
}
|
||||
await budget_limiter.async_log_success_event(kwargs, None, 0, 0)
|
||||
|
||||
for request_model in ("gpt-4", "openai/gpt-4"):
|
||||
with pytest.raises(litellm.BudgetExceededError):
|
||||
await budget_limiter.is_team_within_model_budget(
|
||||
team_id="team-prefixed",
|
||||
team_model_max_budget=team_budget,
|
||||
model=request_model,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -2287,20 +2287,31 @@ class TestScriptModeImportsCanonicalProxyServer:
|
|||
full copy of proxy_server as a top-level module; that shadow copy won the
|
||||
litellm.callbacks registration while uvicorn served the canonical module, so
|
||||
in-memory budget counters were incremented on one instance and read on another
|
||||
and budgets were never enforced without Redis."""
|
||||
and budgets were never enforced without Redis.
|
||||
|
||||
Both module names are pre-stubbed in sys.modules (matching the other run_server
|
||||
tests in this file) so the subprocess never performs the real, heavy proxy_server
|
||||
import: the canonical stub carries the four names run_server pulls, while the
|
||||
top-level stub is empty, so a regression to the bare import fails fast with
|
||||
ImportError instead of timing out."""
|
||||
import subprocess
|
||||
|
||||
repo_root = Path(__file__).resolve().parents[3]
|
||||
driver = (
|
||||
"import sys, runpy\n"
|
||||
"import sys, types, runpy\n"
|
||||
"sys.modules['proxy_server'] = types.ModuleType('proxy_server')\n"
|
||||
"canonical = types.ModuleType('litellm.proxy.proxy_server')\n"
|
||||
"canonical.KeyManagementSettings = object\n"
|
||||
"canonical.ProxyConfig = object\n"
|
||||
"canonical.app = object()\n"
|
||||
"canonical.save_worker_config = lambda **kwargs: None\n"
|
||||
"sys.modules['litellm.proxy.proxy_server'] = canonical\n"
|
||||
"sys.path.insert(0, 'litellm/proxy')\n"
|
||||
"sys.argv = ['proxy_cli.py', '--version']\n"
|
||||
"try:\n"
|
||||
" runpy.run_path('litellm/proxy/proxy_cli.py', run_name='__main__')\n"
|
||||
"except SystemExit:\n"
|
||||
" pass\n"
|
||||
"assert 'litellm.proxy.proxy_server' in sys.modules, 'canonical module not loaded'\n"
|
||||
"assert 'proxy_server' not in sys.modules, 'shadow top-level proxy_server module loaded'\n"
|
||||
"except SystemExit as e:\n"
|
||||
" sys.exit(e.code or 0)\n"
|
||||
)
|
||||
result = subprocess.run(
|
||||
[sys.executable, "-c", driver],
|
||||
|
|
@ -2310,3 +2321,4 @@ class TestScriptModeImportsCanonicalProxyServer:
|
|||
timeout=120,
|
||||
)
|
||||
assert result.returncode == 0, f"stdout={result.stdout}\nstderr={result.stderr}"
|
||||
assert "LiteLLM: Current Version" in result.stdout
|
||||
|
|
|
|||
2
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
2
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -23586,7 +23586,7 @@ export interface components {
|
|||
* @description Default role assigned to new users created
|
||||
* @default internal_user_viewer
|
||||
*/
|
||||
user_role: ("internal_user" | "internal_user_viewer" | "proxy_admin" | "proxy_admin_viewer") | null;
|
||||
user_role: ("proxy_admin" | "proxy_admin_viewer" | "internal_user" | "internal_user_viewer") | null;
|
||||
};
|
||||
/**
|
||||
* DefaultTeamSSOParams
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue