diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 2029948a428..ff4f3293500 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -4522,10 +4522,10 @@ class DefaultInternalUserParams(LiteLLMPydanticObjectBase): user_role: ( Literal[ - LitellmUserRoles.INTERNAL_USER, - LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, + LitellmUserRoles.INTERNAL_USER, + LitellmUserRoles.INTERNAL_USER_VIEW_ONLY, ] | None ) = Field( diff --git a/litellm/proxy/hooks/model_max_budget_limiter.py b/litellm/proxy/hooks/model_max_budget_limiter.py index d4319808510..7a279b20d94 100644 --- a/litellm/proxy/hooks/model_max_budget_limiter.py +++ b/litellm/proxy/hooks/model_max_budget_limiter.py @@ -162,16 +162,24 @@ class _PROXY_VirtualKeyModelMaxBudgetLimiter(RouterBudgetLimiting): self, model: str, internal_model_max_budget: Mapping[str, BudgetConfig] ) -> str | None: """ - Resolve the budget-config entry name `model` matches (exact, then - `{provider}/{model}`-normalized). The entry name is the canonical form for - spend counter keys so prefixed and bare spellings of one model share a counter. + Resolve the budget-config entry name `model` matches: exact first, then with + the `{provider}/` prefix stripped from the request, the entry, or both. The + entry name is the canonical form for spend counter keys so prefixed and bare + spellings of one model share a counter. """ if model in internal_model_max_budget: return model stripped_model = self._get_model_without_custom_llm_provider(model) if stripped_model in internal_model_max_budget: return stripped_model - return None + return next( + ( + entry_name + for entry_name in internal_model_max_budget + if self._get_model_without_custom_llm_provider(entry_name) in (model, stripped_model) + ), + None, + ) def _key_already_covers_model( self, key_model_max_budget: Mapping[str, Mapping[str, str | float]] | None, model: str diff --git a/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py b/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py index b8117ad61c3..cb241a815a8 100644 --- a/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py +++ b/tests/proxy_unit_tests/test_unit_test_max_model_budget_limiter.py @@ -966,3 +966,61 @@ async def test_team_window_start_key_is_per_model_and_duration(budget_limiter): "team_model_budget_start_time:team-windows:gpt-4:1d", "team_model_budget_start_time:team-windows:claude-3:7d", ] + + +def test_get_matched_budget_model_name_strips_entry_prefix(budget_limiter): + """Config entries keyed provider/model must bind bare (and differently prefixed) + requests; before the fix only the request side was normalized, so an + openai/gpt-4 entry never matched a gpt-4 request (Greptile finding).""" + configs = budget_limiter._coerce_budget_configs( + {"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}} + ) + assert ( + budget_limiter._get_matched_budget_model_name( + model="gpt-4", internal_model_max_budget=configs + ) + == "openai/gpt-4" + ) + assert ( + budget_limiter._get_matched_budget_model_name( + model="azure/gpt-4", internal_model_max_budget=configs + ) + == "openai/gpt-4" + ) + assert ( + budget_limiter._get_matched_budget_model_name( + model="claude-3", internal_model_max_budget=configs + ) + is None + ) + + +@pytest.mark.asyncio +async def test_prefixed_team_budget_entry_matches_bare_request(budget_limiter): + """A team cap keyed openai/gpt-4 must govern a bare gpt-4 request end to end: + spend recorded under the canonical entry name, then enforcement raising for + either spelling. Uses the real DualCache, no spend-read mocks.""" + team_budget = {"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}} + kwargs = { + "standard_logging_object": { + "response_cost": 6.0, + "model_group": "gpt-4", + "model": "gpt-4", + "metadata": {"user_api_key_hash": "hash-1"}, + }, + "litellm_params": { + "metadata": { + "user_api_key_team_id": "team-prefixed", + "user_api_key_team_model_max_budget": team_budget, + } + }, + } + await budget_limiter.async_log_success_event(kwargs, None, 0, 0) + + for request_model in ("gpt-4", "openai/gpt-4"): + with pytest.raises(litellm.BudgetExceededError): + await budget_limiter.is_team_within_model_budget( + team_id="team-prefixed", + team_model_max_budget=team_budget, + model=request_model, + ) diff --git a/tests/test_litellm/proxy/test_proxy_cli.py b/tests/test_litellm/proxy/test_proxy_cli.py index 789bf981274..b3bae4503b8 100644 --- a/tests/test_litellm/proxy/test_proxy_cli.py +++ b/tests/test_litellm/proxy/test_proxy_cli.py @@ -2287,20 +2287,31 @@ class TestScriptModeImportsCanonicalProxyServer: full copy of proxy_server as a top-level module; that shadow copy won the litellm.callbacks registration while uvicorn served the canonical module, so in-memory budget counters were incremented on one instance and read on another - and budgets were never enforced without Redis.""" + and budgets were never enforced without Redis. + + Both module names are pre-stubbed in sys.modules (matching the other run_server + tests in this file) so the subprocess never performs the real, heavy proxy_server + import: the canonical stub carries the four names run_server pulls, while the + top-level stub is empty, so a regression to the bare import fails fast with + ImportError instead of timing out.""" import subprocess repo_root = Path(__file__).resolve().parents[3] driver = ( - "import sys, runpy\n" + "import sys, types, runpy\n" + "sys.modules['proxy_server'] = types.ModuleType('proxy_server')\n" + "canonical = types.ModuleType('litellm.proxy.proxy_server')\n" + "canonical.KeyManagementSettings = object\n" + "canonical.ProxyConfig = object\n" + "canonical.app = object()\n" + "canonical.save_worker_config = lambda **kwargs: None\n" + "sys.modules['litellm.proxy.proxy_server'] = canonical\n" "sys.path.insert(0, 'litellm/proxy')\n" "sys.argv = ['proxy_cli.py', '--version']\n" "try:\n" " runpy.run_path('litellm/proxy/proxy_cli.py', run_name='__main__')\n" - "except SystemExit:\n" - " pass\n" - "assert 'litellm.proxy.proxy_server' in sys.modules, 'canonical module not loaded'\n" - "assert 'proxy_server' not in sys.modules, 'shadow top-level proxy_server module loaded'\n" + "except SystemExit as e:\n" + " sys.exit(e.code or 0)\n" ) result = subprocess.run( [sys.executable, "-c", driver], @@ -2310,3 +2321,4 @@ class TestScriptModeImportsCanonicalProxyServer: timeout=120, ) assert result.returncode == 0, f"stdout={result.stdout}\nstderr={result.stderr}" + assert "LiteLLM: Current Version" in result.stdout diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index d9d546e81fe..65ccc94dc5a 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -23586,7 +23586,7 @@ export interface components { * @description Default role assigned to new users created * @default internal_user_viewer */ - user_role: ("internal_user" | "internal_user_viewer" | "proxy_admin" | "proxy_admin_viewer") | null; + user_role: ("proxy_admin" | "proxy_admin_viewer" | "internal_user" | "internal_user_viewer") | null; }; /** * DefaultTeamSSOParams