fix(team): match provider-prefixed team budget entries and stabilize CI

Team model_max_budget entries keyed provider/model (e.g. openai/gpt-4) never
matched bare model-group requests because only the request side was
provider-normalized; the cap was silently unenforced. The matcher now strips
the prefix on both sides and keeps the config entry name as the canonical
spend counter key.

DefaultInternalUserParams declared the same four-role Literal in a different
member order than the three declarations at the top of _types.py; typing's
Literal cache makes the surviving order Python-version dependent, so local and
CI runs generated schema.d.ts with conflicting enum orders. All declarations
now share one order.

The script-mode proxy_cli regression test performed a real proxy_server import
in a subprocess and timed out on CI runners; both module names are now
pre-stubbed in sys.modules (matching the other run_server tests) so the good
path stays fast and a regression to the bare import fails fast with
ImportError.
This commit is contained in:
ryan-crabbe-berri 2026-08-01 23:50:00 -07:00
parent 2abdae078d
commit 5cc8c1b1a2
5 changed files with 91 additions and 13 deletions

View file

@ -4522,10 +4522,10 @@ class DefaultInternalUserParams(LiteLLMPydanticObjectBase):
user_role: (
Literal[
LitellmUserRoles.INTERNAL_USER,
LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
LitellmUserRoles.PROXY_ADMIN,
LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
LitellmUserRoles.INTERNAL_USER,
LitellmUserRoles.INTERNAL_USER_VIEW_ONLY,
]
| None
) = Field(

View file

@ -162,16 +162,24 @@ class _PROXY_VirtualKeyModelMaxBudgetLimiter(RouterBudgetLimiting):
self, model: str, internal_model_max_budget: Mapping[str, BudgetConfig]
) -> str | None:
"""
Resolve the budget-config entry name `model` matches (exact, then
`{provider}/{model}`-normalized). The entry name is the canonical form for
spend counter keys so prefixed and bare spellings of one model share a counter.
Resolve the budget-config entry name `model` matches: exact first, then with
the `{provider}/` prefix stripped from the request, the entry, or both. The
entry name is the canonical form for spend counter keys so prefixed and bare
spellings of one model share a counter.
"""
if model in internal_model_max_budget:
return model
stripped_model = self._get_model_without_custom_llm_provider(model)
if stripped_model in internal_model_max_budget:
return stripped_model
return None
return next(
(
entry_name
for entry_name in internal_model_max_budget
if self._get_model_without_custom_llm_provider(entry_name) in (model, stripped_model)
),
None,
)
def _key_already_covers_model(
self, key_model_max_budget: Mapping[str, Mapping[str, str | float]] | None, model: str

View file

@ -966,3 +966,61 @@ async def test_team_window_start_key_is_per_model_and_duration(budget_limiter):
"team_model_budget_start_time:team-windows:gpt-4:1d",
"team_model_budget_start_time:team-windows:claude-3:7d",
]
def test_get_matched_budget_model_name_strips_entry_prefix(budget_limiter):
"""Config entries keyed provider/model must bind bare (and differently prefixed)
requests; before the fix only the request side was normalized, so an
openai/gpt-4 entry never matched a gpt-4 request (Greptile finding)."""
configs = budget_limiter._coerce_budget_configs(
{"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}}
)
assert (
budget_limiter._get_matched_budget_model_name(
model="gpt-4", internal_model_max_budget=configs
)
== "openai/gpt-4"
)
assert (
budget_limiter._get_matched_budget_model_name(
model="azure/gpt-4", internal_model_max_budget=configs
)
== "openai/gpt-4"
)
assert (
budget_limiter._get_matched_budget_model_name(
model="claude-3", internal_model_max_budget=configs
)
is None
)
@pytest.mark.asyncio
async def test_prefixed_team_budget_entry_matches_bare_request(budget_limiter):
"""A team cap keyed openai/gpt-4 must govern a bare gpt-4 request end to end:
spend recorded under the canonical entry name, then enforcement raising for
either spelling. Uses the real DualCache, no spend-read mocks."""
team_budget = {"openai/gpt-4": {"budget_limit": 5.0, "time_period": "1d"}}
kwargs = {
"standard_logging_object": {
"response_cost": 6.0,
"model_group": "gpt-4",
"model": "gpt-4",
"metadata": {"user_api_key_hash": "hash-1"},
},
"litellm_params": {
"metadata": {
"user_api_key_team_id": "team-prefixed",
"user_api_key_team_model_max_budget": team_budget,
}
},
}
await budget_limiter.async_log_success_event(kwargs, None, 0, 0)
for request_model in ("gpt-4", "openai/gpt-4"):
with pytest.raises(litellm.BudgetExceededError):
await budget_limiter.is_team_within_model_budget(
team_id="team-prefixed",
team_model_max_budget=team_budget,
model=request_model,
)

View file

@ -2287,20 +2287,31 @@ class TestScriptModeImportsCanonicalProxyServer:
full copy of proxy_server as a top-level module; that shadow copy won the
litellm.callbacks registration while uvicorn served the canonical module, so
in-memory budget counters were incremented on one instance and read on another
and budgets were never enforced without Redis."""
and budgets were never enforced without Redis.
Both module names are pre-stubbed in sys.modules (matching the other run_server
tests in this file) so the subprocess never performs the real, heavy proxy_server
import: the canonical stub carries the four names run_server pulls, while the
top-level stub is empty, so a regression to the bare import fails fast with
ImportError instead of timing out."""
import subprocess
repo_root = Path(__file__).resolve().parents[3]
driver = (
"import sys, runpy\n"
"import sys, types, runpy\n"
"sys.modules['proxy_server'] = types.ModuleType('proxy_server')\n"
"canonical = types.ModuleType('litellm.proxy.proxy_server')\n"
"canonical.KeyManagementSettings = object\n"
"canonical.ProxyConfig = object\n"
"canonical.app = object()\n"
"canonical.save_worker_config = lambda **kwargs: None\n"
"sys.modules['litellm.proxy.proxy_server'] = canonical\n"
"sys.path.insert(0, 'litellm/proxy')\n"
"sys.argv = ['proxy_cli.py', '--version']\n"
"try:\n"
" runpy.run_path('litellm/proxy/proxy_cli.py', run_name='__main__')\n"
"except SystemExit:\n"
" pass\n"
"assert 'litellm.proxy.proxy_server' in sys.modules, 'canonical module not loaded'\n"
"assert 'proxy_server' not in sys.modules, 'shadow top-level proxy_server module loaded'\n"
"except SystemExit as e:\n"
" sys.exit(e.code or 0)\n"
)
result = subprocess.run(
[sys.executable, "-c", driver],
@ -2310,3 +2321,4 @@ class TestScriptModeImportsCanonicalProxyServer:
timeout=120,
)
assert result.returncode == 0, f"stdout={result.stdout}\nstderr={result.stderr}"
assert "LiteLLM: Current Version" in result.stdout

View file

@ -23586,7 +23586,7 @@ export interface components {
* @description Default role assigned to new users created
* @default internal_user_viewer
*/
user_role: ("internal_user" | "internal_user_viewer" | "proxy_admin" | "proxy_admin_viewer") | null;
user_role: ("proxy_admin" | "proxy_admin_viewer" | "internal_user" | "internal_user_viewer") | null;
};
/**
* DefaultTeamSSOParams