mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(rate-limit): resolve router model_name aliases to real provider (#27914)
* fix(rate-limit): resolve router model_name aliases to real provider
For nearly every real LiteLLM proxy deployment the request model is a
router model_name alias (e.g. 'tpm-locked' -> litellm_params.model:
openai/gpt-4o-mini), and 'litellm.get_llm_provider' doesn't know about
router aliases — it raises 'LLMProviderNotProvidedError'. The resolver
then fell through to the defensive 'litellm_proxy' fallback, so the
'llm_provider' field this PR adds was effectively always
'litellm_proxy' in the field, defeating its purpose for the most common
proxy configuration.
Add a router-alias fallback step: when 'get_llm_provider' raises, scan
the active 'llm_router.model_list' for a deployment whose 'model_name'
matches the request model and resolve from its 'litellm_params.model'
instead. If multiple deployments share the same alias (load-balancing
case) the first one wins — every deployment under one alias should
agree on provider in any sensible config, and 'first' is deterministic
so the Prometheus label stays stable.
Defensive throughout: an uninitialized router, a malformed deployment,
a 'litellm_params.model' that itself fails 'get_llm_provider' — every
branch falls through to the existing 'litellm_proxy' fallback rather
than letting a secondary exception escape and mask the rate-limit
error we're trying to surface.
Tests:
- test_router_alias_resolves_to_underlying_provider: alias
'tpm-locked' -> 'openai/gpt-4o-mini' produces provider='openai',
model='gpt-4o-mini'.
- test_router_alias_with_multiple_deployments_uses_first.
- test_router_alias_unknown_falls_back.
- test_router_alias_with_malformed_deployment_falls_back.
- Existing fallback test updated to also stub
'litellm.proxy.proxy_server.llm_router' so it exercises the
full 'no resolution anywhere' path.
Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
* fix(rate-limit): harden router alias resolver + test isolation
- Wrap _resolve_provider_from_router_alias loop in top-level try/except so
a non-iterable model_list / unexpected deployment shape can't escape and
mask the 429 with a 500.
- Type-check litellm_params before .get() to handle non-dict truthy values.
- Patch llm_router=None in the parametrized fallback test so a router left
by another test in the session can't redirect the unknown-model path.
---------
Co-authored-by: Cursor Agent <cursoragent@cursor.com>
Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
f2d324310a
commit
7747ad050f
2 changed files with 210 additions and 7 deletions
|
|
@ -26,11 +26,21 @@ def resolve_llm_provider_for_rate_limit(
|
|||
``litellm_proxy_failed_requests_metric`` show up with
|
||||
``exception_class="RateLimitError"`` and no provider attribution.
|
||||
|
||||
Wrapped defensively: if ``model`` is missing, malformed, or
|
||||
``get_llm_provider`` raises (unknown alias, router-only model, etc.) we
|
||||
fall back to ``("", "litellm_proxy")`` so we never break the request path
|
||||
by piling a second exception on top of the rate-limit one we're trying to
|
||||
raise.
|
||||
Resolution order:
|
||||
|
||||
1. ``litellm.get_llm_provider(model)`` — covers raw provider/model
|
||||
strings the SDK already understands (``"gpt-4o-mini"``,
|
||||
``"anthropic/claude-3-5-sonnet"``, ``"bedrock/..."`` etc.).
|
||||
2. **Router alias fallback** — nearly every real proxy deployment
|
||||
routes through a router ``model_name`` alias (e.g.
|
||||
``"tpm-locked"`` → ``litellm_params.model: openai/gpt-4o-mini``).
|
||||
``get_llm_provider`` doesn't know router aliases, so without this
|
||||
step every alias call ended up labeled ``"litellm_proxy"``,
|
||||
defeating the field's purpose for the most common case.
|
||||
3. Defensive fallback to ``("", "litellm_proxy")`` — used only when
|
||||
``model`` is missing, malformed, or both lookups fail. We never let
|
||||
a secondary exception escape and mask the rate-limit error we're
|
||||
trying to surface.
|
||||
"""
|
||||
if not model:
|
||||
return "", PROXY_LLM_PROVIDER_FALLBACK
|
||||
|
|
@ -43,6 +53,9 @@ def resolve_llm_provider_for_rate_limit(
|
|||
custom_llm_provider or PROXY_LLM_PROVIDER_FALLBACK,
|
||||
)
|
||||
except Exception as e:
|
||||
alias_resolution = _resolve_provider_from_router_alias(model)
|
||||
if alias_resolution is not None:
|
||||
return alias_resolution
|
||||
verbose_proxy_logger.debug(
|
||||
"rate_limiter_utils.resolve_llm_provider_for_rate_limit: "
|
||||
"could not resolve provider for model=%s, falling back to %s. err=%s",
|
||||
|
|
@ -53,6 +66,60 @@ def resolve_llm_provider_for_rate_limit(
|
|||
return model, PROXY_LLM_PROVIDER_FALLBACK
|
||||
|
||||
|
||||
def _resolve_provider_from_router_alias(
|
||||
model: str,
|
||||
) -> Optional[Tuple[str, str]]:
|
||||
"""
|
||||
Resolve a router ``model_name`` alias to ``(underlying_model, provider)``
|
||||
by scanning the active router's ``model_list``.
|
||||
|
||||
Returns ``None`` if the router isn't initialized, the alias isn't
|
||||
registered, the deployment has no usable ``litellm_params.model``, or
|
||||
any underlying lookup raises. Callers fall through to the defensive
|
||||
``litellm_proxy`` fallback in that case — never raising secondary
|
||||
exceptions out of the rate-limit raise path.
|
||||
"""
|
||||
try:
|
||||
from litellm.proxy.proxy_server import llm_router
|
||||
except Exception:
|
||||
return None
|
||||
if llm_router is None:
|
||||
return None
|
||||
try:
|
||||
model_list = getattr(llm_router, "model_list", None)
|
||||
if not model_list:
|
||||
return None
|
||||
for deployment in model_list:
|
||||
if not isinstance(deployment, dict):
|
||||
continue
|
||||
if deployment.get("model_name") != model:
|
||||
continue
|
||||
params = deployment.get("litellm_params")
|
||||
if not isinstance(params, dict):
|
||||
continue
|
||||
underlying_model = params.get("model")
|
||||
if not isinstance(underlying_model, str) or not underlying_model:
|
||||
continue
|
||||
try:
|
||||
resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
model=underlying_model,
|
||||
)
|
||||
except Exception:
|
||||
continue
|
||||
if not custom_llm_provider:
|
||||
continue
|
||||
# Prefer the underlying provider-qualified model so the failure
|
||||
# callback / Prometheus label points at the actual deployment, not
|
||||
# the alias.
|
||||
return (
|
||||
resolved_model or underlying_model,
|
||||
custom_llm_provider,
|
||||
)
|
||||
return None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def convert_priority_to_percent(
|
||||
value: Union[float, PriorityReservationDict], model_info: Optional[ModelGroupInfo]
|
||||
) -> float:
|
||||
|
|
|
|||
|
|
@ -141,7 +141,10 @@ class TestResolveLLMProviderForRateLimit:
|
|||
# Must never raise — the resolver wraps `get_llm_provider` defensively
|
||||
# because raising here would mask the rate-limit error we're trying
|
||||
# to surface to the user.
|
||||
resolved_model, provider = resolve_llm_provider_for_rate_limit(model)
|
||||
# Pin llm_router to None so the alias-fallback path doesn't pick up
|
||||
# a router left behind by another test in the session.
|
||||
with patch("litellm.proxy.proxy_server.llm_router", None):
|
||||
resolved_model, provider = resolve_llm_provider_for_rate_limit(model)
|
||||
assert provider == PROXY_LLM_PROVIDER_FALLBACK
|
||||
# Resolver returns the input model verbatim on the unknown branch so
|
||||
# the `.model` attribute is never silently swapped to a different one.
|
||||
|
|
@ -153,15 +156,148 @@ class TestResolveLLMProviderForRateLimit:
|
|||
def test_get_llm_provider_raising_is_swallowed(self):
|
||||
# If get_llm_provider itself blows up (unexpected error), we still
|
||||
# fall back rather than letting the secondary exception escape.
|
||||
# No router is registered in this test, so the alias-fallback path
|
||||
# also yields None and we land at PROXY_LLM_PROVIDER_FALLBACK.
|
||||
with patch.object(
|
||||
litellm,
|
||||
"get_llm_provider",
|
||||
side_effect=RuntimeError("boom"),
|
||||
):
|
||||
resolved_model, provider = resolve_llm_provider_for_rate_limit("anything")
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.llm_router",
|
||||
None,
|
||||
):
|
||||
resolved_model, provider = resolve_llm_provider_for_rate_limit(
|
||||
"anything"
|
||||
)
|
||||
assert provider == PROXY_LLM_PROVIDER_FALLBACK
|
||||
assert resolved_model == "anything"
|
||||
|
||||
def test_router_alias_resolves_to_underlying_provider(self):
|
||||
"""
|
||||
Nearly every real LiteLLM proxy deployment uses router aliases:
|
||||
|
||||
model_list:
|
||||
- model_name: tpm-locked
|
||||
litellm_params:
|
||||
model: openai/gpt-4o-mini
|
||||
...
|
||||
|
||||
``litellm.get_llm_provider("tpm-locked")`` doesn't know about
|
||||
router aliases and raises. Before this fix the resolver fell
|
||||
through to ``"litellm_proxy"``, defeating the whole point of the
|
||||
``llm_provider`` field on the rate-limit error. The alias path
|
||||
must look the deployment up in the router's ``model_list`` and
|
||||
resolve from its ``litellm_params.model``.
|
||||
"""
|
||||
|
||||
class _FakeRouter:
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "tpm-locked",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o-mini",
|
||||
"api_key": "fake",
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.llm_router",
|
||||
_FakeRouter(),
|
||||
):
|
||||
resolved_model, provider = resolve_llm_provider_for_rate_limit("tpm-locked")
|
||||
assert provider == "openai", (
|
||||
f"Router-alias path must resolve through litellm_params.model, "
|
||||
f"not fall through to {PROXY_LLM_PROVIDER_FALLBACK!r}. Got "
|
||||
f"provider={provider!r}, model={resolved_model!r}."
|
||||
)
|
||||
# The resolved model should point at the underlying deployment so
|
||||
# downstream Prometheus labels / failure callbacks attribute the
|
||||
# 429 to the real upstream, not the alias.
|
||||
assert resolved_model == "gpt-4o-mini"
|
||||
|
||||
def test_router_alias_with_multiple_deployments_uses_first(self):
|
||||
"""
|
||||
When an alias maps to multiple deployments (the load-balancing
|
||||
case), the rate-limit error fired at the *alias* level is
|
||||
deployment-agnostic — we have no way of knowing which one would
|
||||
have been picked. Use the first deployment's underlying provider:
|
||||
every deployment under one alias should agree on provider in any
|
||||
sensible config, and 'first' is deterministic so the Prometheus
|
||||
label is stable.
|
||||
"""
|
||||
|
||||
class _FakeRouter:
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "claude-pool",
|
||||
"litellm_params": {"model": "anthropic/claude-3-5-sonnet"},
|
||||
},
|
||||
{
|
||||
"model_name": "claude-pool",
|
||||
"litellm_params": {"model": "anthropic/claude-3-5-haiku"},
|
||||
},
|
||||
]
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.llm_router",
|
||||
_FakeRouter(),
|
||||
):
|
||||
_, provider = resolve_llm_provider_for_rate_limit("claude-pool")
|
||||
assert provider == "anthropic"
|
||||
|
||||
def test_router_alias_unknown_falls_back(self):
|
||||
"""
|
||||
Alias not in the router model_list — both lookups fail, so we
|
||||
land at the defensive ``litellm_proxy`` fallback rather than
|
||||
raising.
|
||||
"""
|
||||
|
||||
class _FakeRouter:
|
||||
model_list = [
|
||||
{
|
||||
"model_name": "tpm-locked",
|
||||
"litellm_params": {"model": "openai/gpt-4o-mini"},
|
||||
}
|
||||
]
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.llm_router",
|
||||
_FakeRouter(),
|
||||
):
|
||||
resolved_model, provider = resolve_llm_provider_for_rate_limit(
|
||||
"not-an-alias"
|
||||
)
|
||||
assert provider == PROXY_LLM_PROVIDER_FALLBACK
|
||||
assert resolved_model == "not-an-alias"
|
||||
|
||||
def test_router_alias_with_malformed_deployment_falls_back(self):
|
||||
"""
|
||||
A deployment in the router model_list with no usable
|
||||
``litellm_params.model`` (or where ``get_llm_provider`` on the
|
||||
underlying string also raises) must not crash the resolver —
|
||||
fall through to the defensive fallback.
|
||||
"""
|
||||
|
||||
class _FakeRouter:
|
||||
model_list = [
|
||||
{"model_name": "broken", "litellm_params": {}},
|
||||
{"model_name": "broken", "litellm_params": {"model": ""}},
|
||||
{
|
||||
"model_name": "broken",
|
||||
"litellm_params": {"model": "nonsense-no-provider"},
|
||||
},
|
||||
]
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.llm_router",
|
||||
_FakeRouter(),
|
||||
):
|
||||
resolved_model, provider = resolve_llm_provider_for_rate_limit("broken")
|
||||
assert provider == PROXY_LLM_PROVIDER_FALLBACK
|
||||
assert resolved_model == "broken"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# parallel_request_limiter v1
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue