From 5cc8c1b1a22635bdd81710365b9d461d6590ef3e Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Sat, 1 Aug 2026 23:50:00 -0700 Subject: [PATCH] fix(team): match provider-prefixed team budget entries and stabilize CI Team model_max_budget entries keyed provider/model (e.g. openai/gpt-4) never matched bare model-group requests because only the request side was provider-normalized; the cap was silently unenforced. The matcher now strips the prefix on both sides and keeps the config entry name as the canonical spend counter key. DefaultInternalUserParams declared the same four-role Literal in a different member order than the three declarations at the top of _types.py; typing's Literal cache makes the surviving order Python-version dependent, so local and CI runs generated schema.d.ts with conflicting enum orders. All declarations now share one order. The script-mode proxy_cli regression test performed a real proxy_server import in a subprocess and timed out on CI runners; both module names are now pre-stubbed in sys.modules (matching the other run_server tests) so the good path stays fast and a regression to the bare import fails fast with ImportError. --- litellm/proxy/_types.py | 4 +- .../proxy/hooks/model_max_budget_limiter.py | 16 +++-- ...test_unit_test_max_model_budget_limiter.py | 58 +++++++++++++++++++ tests/test_litellm/proxy/test_proxy_cli.py | 24 ++++++-- ui/litellm-dashboard/src/lib/http/schema.d.ts | 2 +- 5 files changed, 91 insertions(+), 13 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 2029948a428..ff4f3293500 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -4522,10 +4522,10 @@ class DefaultInternalUserParams(LiteLLMPydanticObjectBase): user_role: ( Literal[ - LitellmUserRoles.INTERNAL_USER, - LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, + LitellmUserRoles.INTERNAL_USER, + LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, ] | None ) = Field( diff --git a/litellm/proxy/hooks/model_max_budget_limiter.py b/litellm/proxy/hooks/model_max_budget_limiter.py index d4319808510..7a279b20d94 100644 --- a/litellm/proxy/hooks/model_max_budget_limiter.py +++ b/litellm/proxy/hooks/model_max_budget_limiter.py @@ -162,16 +162,24 @@ class _PROXY_VirtualKeyModelMaxBudgetLimiter(RouterBudgetLimiting): self, model: str, internal_model_max_budget: Mapping[str, BudgetConfig] ) -> str | None: """ - Resolve the budget-config entry name `model` matches (exact, then - `{provider}/{model}`-normalized). The entry name is the canonical form for - spend counter keys so prefixed and bare spellings of one model share a counter. + Resolve the budget-config entry name `model` matches: exact first, then with + the `{provider}/` prefix stripped from the request, the entry, or both. The + entry name is the canonical form for spend counter keys so prefixed and bare + spellings of one model share a counter. """ if model in internal_model_max_budget: return model stripped_model = self._get_model_without_custom_llm_provider(model) if stripped_model in internal_model_max_budget: return stripped_model - return None + return next( + ( + entry_name + for entry_name in internal_model_max_budget + if self._get_model_without_custom_llm_provider(entry_name) in (model, stripped_model) + ), + None, + ) def _key_already_covers_model( self, key_model_max_budget: Mapping[str, Mapping[str, str | float]] | None, model: str diff --git a/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py b/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py index b8117ad61c3..cb241a815a8 100644 --- a/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py +++ b/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py @@ -966,3 +966,61 @@ async def test_team_window_start_key_is_per_model_and_duration(budget_limiter): "team_model_budget_start_time:team-windows:gpt-4:1d", "team_model_budget_start_time:team-windows:claude-3:7d", ] + + +def test_get_matched_budget_model_name_strips_entry_prefix(budget_limiter): + """Config entries keyed provider/model must bind bare (and differently prefixed) + requests; before the fix only the request side was normalized, so an + openai/gpt-4 entry never matched a gpt-4 request (Greptile finding).""" + configs = budget_limiter._coerce_budget_configs( + {"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}} + ) + assert ( + budget_limiter._get_matched_budget_model_name( + model="gpt-4", internal_model_max_budget=configs + ) + == "openai/gpt-4" + ) + assert ( + budget_limiter._get_matched_budget_model_name( + model="azure/gpt-4", internal_model_max_budget=configs + ) + == "openai/gpt-4" + ) + assert ( + budget_limiter._get_matched_budget_model_name( + model="claude-3", internal_model_max_budget=configs + ) + is None + ) + + +@pytest.mark.asyncio +async def test_prefixed_team_budget_entry_matches_bare_request(budget_limiter): + """A team cap keyed openai/gpt-4 must govern a bare gpt-4 request end to end: + spend recorded under the canonical entry name, then enforcement raising for + either spelling. Uses the real DualCache, no spend-read mocks.""" + team_budget = {"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}} + kwargs = { + "standard_logging_object": { + "response_cost": 6.0, + "model_group": "gpt-4", + "model": "gpt-4", + "metadata": {"user_api_key_hash": "hash-1"}, + }, + "litellm_params": { + "metadata": { + "user_api_key_team_id": "team-prefixed", + "user_api_key_team_model_max_budget": team_budget, + } + }, + } + await budget_limiter.async_log_success_event(kwargs, None, 0, 0) + + for request_model in ("gpt-4", "openai/gpt-4"): + with pytest.raises(litellm.BudgetExceededError): + await budget_limiter.is_team_within_model_budget( + team_id="team-prefixed", + team_model_max_budget=team_budget, + model=request_model, + ) diff --git a/tests/test_litellm/proxy/test_proxy_cli.py b/tests/test_litellm/proxy/test_proxy_cli.py index 789bf981274..b3bae4503b8 100644 --- a/tests/test_litellm/proxy/test_proxy_cli.py +++ b/tests/test_litellm/proxy/test_proxy_cli.py @@ -2287,20 +2287,31 @@ class TestScriptModeImportsCanonicalProxyServer: full copy of proxy_server as a top-level module; that shadow copy won the litellm.callbacks registration while uvicorn served the canonical module, so in-memory budget counters were incremented on one instance and read on another - and budgets were never enforced without Redis.""" + and budgets were never enforced without Redis. + + Both module names are pre-stubbed in sys.modules (matching the other run_server + tests in this file) so the subprocess never performs the real, heavy proxy_server + import: the canonical stub carries the four names run_server pulls, while the + top-level stub is empty, so a regression to the bare import fails fast with + ImportError instead of timing out.""" import subprocess repo_root = Path(__file__).resolve().parents[3] driver = ( - "import sys, runpy\n" + "import sys, types, runpy\n" + "sys.modules['proxy_server'] = types.ModuleType('proxy_server')\n" + "canonical = types.ModuleType('litellm.proxy.proxy_server')\n" + "canonical.KeyManagementSettings = object\n" + "canonical.ProxyConfig = object\n" + "canonical.app = object()\n" + "canonical.save_worker_config = lambda **kwargs: None\n" + "sys.modules['litellm.proxy.proxy_server'] = canonical\n" "sys.path.insert(0, 'litellm/proxy')\n" "sys.argv = ['proxy_cli.py', '--version']\n" "try:\n" " runpy.run_path('litellm/proxy/proxy_cli.py', run_name='__main__')\n" - "except SystemExit:\n" - " pass\n" - "assert 'litellm.proxy.proxy_server' in sys.modules, 'canonical module not loaded'\n" - "assert 'proxy_server' not in sys.modules, 'shadow top-level proxy_server module loaded'\n" + "except SystemExit as e:\n" + " sys.exit(e.code or 0)\n" ) result = subprocess.run( [sys.executable, "-c", driver], @@ -2310,3 +2321,4 @@ class TestScriptModeImportsCanonicalProxyServer: timeout=120, ) assert result.returncode == 0, f"stdout={result.stdout}\nstderr={result.stderr}" + assert "LiteLLM: Current Version" in result.stdout diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index d9d546e81fe..65ccc94dc5a 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -23586,7 +23586,7 @@ export interface components { * @description Default role assigned to new users created * @default internal_user_viewer */ - user_role: ("internal_user" | "internal_user_viewer" | "proxy_admin" | "proxy_admin_viewer") | null; + user_role: ("proxy_admin" | "proxy_admin_viewer" | "internal_user" | "internal_user_viewer") | null; }; /** * DefaultTeamSSOParams