litellm/tests/unit/proxy/proxy_server/test_routes_models.py
devin-ai-integration[bot] fe683ea139
feat(proxy): serve a Codex-native model catalog with per-model service_tiers from /v1/models (#44136)
* feat(proxy): serve a Codex-native model catalog with per-model service_tiers from /v1/models

GET /v1/models and /models answer Codex CLI's catalog fetch (the request
carrying its client_version query parameter) with Codex's own
{"models": [...]} shape: a model Codex knows keeps the metadata of its
bundled 0.159.3 catalog (vendored), any other model gets Codex's fallback
entry, and model_info.service_tiers becomes each entry's service tiers so
Codex offers them as slash commands that send service_tier upstream.
Without the parameter the OpenAI list shape is unchanged. The CLI's
litellm agents codex catalog shares the same builder.

* fix(proxy): offer a Codex service tier only when every deployment of the model lists it

* fix(codex-catalog): an invalid service_tiers value offers no tier for the model

* fix(codex-catalog): read service tiers off the deployments the key's team can route to

A tier is offered to Codex only when every deployment of the model name a
request from the key's team can route to lists it, so another team's
deployment of the name and a deployment an admin paused via model_info.blocked
no longer withhold or add tiers for requests that never reach them

The catalog's always-null fields are annotated NoneType so the module imports
under pydantic 2.12.0 on Python 3.14, the lowest pin the MCP resolve job
installs, which rejects a None annotation with a None default

* test(codex-catalog): drop the redundant module docstring and sort the imports

* test(integration): add the Codex catalog audit cells and the multi-worker convergence note

* test(integration): clean up every catalog test model and answer the refresh GET

* fix(proxy): keep tiered models under Codex's catalog cut and resolve alias tiers

Under Codex's 1 MiB catalog limit the entries offering a service tier are kept
ahead of those offering none, each group in model_list order, with every kept
entry at its listing position, so the model an operator configured tiers for
survives a wide key's long listing. A model_group_alias row reads its target's
deployments, so it carries the target's tiers and stock metadata under the
alias name.

* fix(proxy): pick Codex catalog metadata per team and skip entries too large for the cut

The upstream model that selects Codex's stock entry was read off the first deployment of a name
without checking the key's team, so a team whose requests route to a different deployment could be
handed another team's prompt, reasoning levels, and tiers. The upstream model and the tiers now come
from the same team-aware selection routing uses, and a caller with no team reads the deployments no
team owns

The byte cut kept a prefix of the tier-first order, so one entry larger than the whole limit emptied
the catalog. An entry too large for the bytes left is now passed over and the smaller ones after it
are still kept

---------

Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
2026-10-03 20:42:55 +00:00

516 lines
21 KiB
Python

"""Behavior pins for ``proxy_server.py`` model routes.
Pins (PR2):
- GET /v1/models
- GET /models
- GET /v1/models/{model_id}
- GET /models/{model_id}
"""
from __future__ import annotations
from unittest.mock import MagicMock
import pytest
import litellm
from litellm.proxy import proxy_server
from litellm.proxy import utils as proxy_utils
from litellm.proxy.utils import create_model_info_response
from litellm.types.router import DeploymentModelListingInfo
from .conftest import normalize # type: ignore[import-not-found]
def _stub_model_info_response(
model_id: str = "gpt-4", provider: str = "openai"
) -> dict:
return {
"id": model_id,
"object": "model",
"created": 0,
"owned_by": provider,
}
@pytest.fixture
def patched_models(monkeypatch):
"""Stub router + utility helpers used by the /models routes."""
from litellm.proxy import utils as proxy_utils
router = MagicMock()
router.get_fully_blocked_model_names = MagicMock(return_value=set())
router.get_model_names = MagicMock(return_value=["gpt-4", "claude-sonnet"])
router.get_model_access_groups = MagicMock(return_value={})
deployment = MagicMock()
deployment.litellm_params.model = "gpt-4"
router.get_deployment_by_model_group_name = MagicMock(return_value=deployment)
router.get_routable_upstream_model = MagicMock(return_value="gpt-4")
router.get_configured_display_name = MagicMock(return_value=None)
router.get_configured_service_tiers = MagicMock(return_value=())
monkeypatch.setattr(proxy_server, "llm_router", router)
monkeypatch.setattr(proxy_server, "prisma_client", MagicMock())
async def _fake_get_available_models_for_user(**kwargs):
return ["gpt-4", "claude-sonnet"]
monkeypatch.setattr(
proxy_utils,
"get_available_models_for_user",
_fake_get_available_models_for_user,
)
def _fake_create_model_info_response(model_id, provider="openai", **kwargs):
return _stub_model_info_response(model_id=model_id, provider=provider)
monkeypatch.setattr(
proxy_utils, "create_model_info_response", _fake_create_model_info_response
)
monkeypatch.setattr(proxy_utils, "validate_model_access", lambda **kwargs: None)
monkeypatch.setattr(
litellm,
"get_llm_provider",
lambda model: (model, "openai", None, None),
)
return router
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_get_models_happy_path(client, auth_as, patched_models, path):
"""Pins: ``GET /v1/models``, ``GET /models``."""
with auth_as():
response = client.get(path)
assert response.status_code == 200
assert normalize(response.json()) == {
"data": [
{
"id": "<VOLATILE>",
"object": "model",
"created": "<VOLATILE>",
"owned_by": "openai",
},
{
"id": "<VOLATILE>",
"object": "model",
"created": "<VOLATILE>",
"owned_by": "openai",
},
],
"object": "list",
}
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_get_models_anthropic_format_when_header_present(
client, auth_as, patched_models, path
):
"""Pins: ``GET /v1/models`` returns the Anthropic-native models shape when
the caller sends an ``anthropic-version`` header (Claude Code gateway
discovery), while the default OpenAI shape is unchanged without it."""
with auth_as():
response = client.get(path, headers={"anthropic-version": "2023-06-01"})
assert response.status_code == 200
body = response.json()
assert "object" not in body
assert body["has_more"] is False
assert body["first_id"] == "gpt-4"
assert body["last_id"] == "claude-sonnet"
assert [m["id"] for m in body["data"]] == ["gpt-4", "claude-sonnet"]
for entry in body["data"]:
assert entry["type"] == "model"
assert entry["display_name"] == entry["id"]
assert entry["created_at"].endswith("Z")
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_anthropic_format_exposes_token_limits(
client, auth_as, patched_models, monkeypatch, path
):
"""Claude Code sizes requests off the listing, so the Anthropic-native entries
carry the same token limits the OpenAI listing resolves, with the output budget
named max_tokens as the Messages API names it."""
from litellm.proxy import utils as proxy_utils
def _create_model_info_response(model_id, provider="openai", **kwargs):
if model_id != "claude-sonnet":
return _stub_model_info_response(model_id=model_id, provider=provider)
return {
**_stub_model_info_response(model_id=model_id, provider=provider),
"max_input_tokens": 200000,
"max_output_tokens": 64000,
}
monkeypatch.setattr(
proxy_utils, "create_model_info_response", _create_model_info_response
)
with auth_as():
response = client.get(path, headers={"anthropic-version": "2023-06-01"})
assert response.status_code == 200
gpt_4, claude = response.json()["data"]
assert claude["max_input_tokens"] == 200000
assert claude["max_tokens"] == 64000
assert "max_output_tokens" not in claude
assert gpt_4["max_input_tokens"] is None
assert gpt_4["max_tokens"] is None
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_anthropic_format_carries_router_configured_token_limits(client, auth_as, patched_models, monkeypatch, path):
"""Pins the whole resolution chain, not just the formatter: a deployment's
configured limits beat the cost map, and the configured output budget is what
lands on the Anthropic ``max_tokens``. All eight limits differ, so an entry
built from another entry's lookup shows up as the wrong numbers."""
def _configured(model_name):
max_input, max_output = (300000, 32000) if model_name == "gpt-4" else (500000, 4096)
return DeploymentModelListingInfo(
cost_map_keys=(model_name,), max_input_tokens=max_input, max_output_tokens=max_output
)
def _cost_map_lookup(model_id):
max_input, max_output = (200000, 64000) if model_id == "gpt-4" else (100000, 8000)
return {"max_input_tokens": max_input, "max_output_tokens": max_output, "mode": "chat"}
patched_models.get_model_listing_info = MagicMock(side_effect=_configured)
def _resolved(**kwargs):
return create_model_info_response(**kwargs, get_model_info=_cost_map_lookup)
monkeypatch.setattr(proxy_utils, "create_model_info_response", _resolved)
with auth_as():
response = client.get(path, headers={"anthropic-version": "2023-06-01"})
assert response.status_code == 200
gpt_4, claude = response.json()["data"]
assert (gpt_4["max_input_tokens"], gpt_4["max_tokens"]) == (300000, 32000)
assert (claude["max_input_tokens"], claude["max_tokens"]) == (500000, 4096)
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_anthropic_format_uses_configured_display_name(client, auth_as, patched_models, path):
"""A deployment's ``model_info.display_name`` becomes the Anthropic-native
``display_name`` so Claude Code's picker shows a clean name while the id keeps
routing; models without one keep the id fallback, and the OpenAI-shaped
listing carries no display_name either way."""
def _configured(model_name):
return "Kimi K3" if model_name == "gpt-4" else None
patched_models.get_configured_display_name = MagicMock(side_effect=_configured)
with auth_as():
anthropic_response = client.get(path, headers={"anthropic-version": "2023-06-01"})
openai_response = client.get(path)
assert anthropic_response.status_code == 200
gpt_4, claude = anthropic_response.json()["data"]
assert (gpt_4["id"], gpt_4["display_name"]) == ("gpt-4", "Kimi K3")
assert (claude["id"], claude["display_name"]) == ("claude-sonnet", "claude-sonnet")
assert openai_response.status_code == 200
openai_models = openai_response.json()["data"]
assert [m["id"] for m in openai_models] == ["gpt-4", "claude-sonnet"]
assert all("display_name" not in m for m in openai_models)
@pytest.mark.parametrize("params", [{}, {"scope": "expand"}])
def test_anthropic_display_name_resolved_via_internal_team_key(
client, auth_as, patched_models, monkeypatch, params
):
"""For a team-scoped row the configured display name must be looked up by the
internal routing key while the entry itself is keyed by the public name, so
the clean name lands on the id the client actually sees."""
from litellm.proxy import utils as proxy_utils
from litellm.proxy.auth import model_checks
internal_name = "model_name_team-1_c0ffee"
patched_models.get_model_list = MagicMock(
return_value=[
{
"model_name": internal_name,
"model_info": {
"team_id": "team-1",
"team_public_model_name": "gpt-4-team",
},
}
]
)
patched_models.get_model_names = MagicMock(return_value=[internal_name])
patched_models.get_configured_display_name = MagicMock(
side_effect=lambda model_name: "Team GPT" if model_name == internal_name else None
)
async def _fake_get_available_models_for_user(**kwargs):
return [internal_name]
monkeypatch.setattr(
proxy_utils,
"get_available_models_for_user",
_fake_get_available_models_for_user,
)
monkeypatch.setattr(
model_checks, "get_complete_model_list", lambda **kwargs: [internal_name]
)
with auth_as():
response = client.get(
"/v1/models", params=params, headers={"anthropic-version": "2023-06-01"}
)
assert response.status_code == 200
(entry,) = response.json()["data"]
assert (entry["id"], entry["display_name"]) == ("gpt-4-team", "Team GPT")
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_get_models_invalid_scope_returns_400(client, auth_as, patched_models, path):
"""Pins: ``GET /v1/models``, ``GET /models`` (error path: invalid scope)."""
with auth_as():
response = client.get(path, params={"scope": "not-a-real-scope"})
assert response.status_code == 400
assert "Invalid scope parameter" in str(response.json())
@pytest.mark.parametrize("path", ["/v1/models/gpt-4", "/models/gpt-4"])
def test_get_model_by_id_happy_path(client, auth_as, patched_models, path):
"""Pins: ``GET /v1/models/{model_id}``, ``GET /models/{model_id}``."""
with auth_as():
response = client.get(path)
assert response.status_code == 200
assert normalize(response.json()) == {
"id": "<VOLATILE>",
"object": "model",
"created": "<VOLATILE>",
"owned_by": "openai",
}
@pytest.mark.parametrize("path", ["/v1/models/missing", "/models/missing"])
def test_get_model_by_id_not_found(client, auth_as, patched_models, path):
"""Pins: ``GET /v1/models/{model_id}``, ``GET /models/{model_id}`` (error: 404)."""
patched_models.get_deployment_by_model_group_name = MagicMock(return_value=None)
with auth_as():
response = client.get(path)
assert response.status_code == 404
assert "not found" in response.text.lower()
@pytest.mark.parametrize("params", [{}, {"scope": "expand"}])
def test_anthropic_format_returns_public_team_model_name(
client, auth_as, patched_models, monkeypatch, params
):
"""Regression: the Anthropic-native listing must go through the same team
name translation as the OpenAI listing, so a caller never sees the internal
``model_name_{team_id}_{uuid}`` routing key."""
from litellm.proxy import utils as proxy_utils
from litellm.proxy.auth import model_checks
internal_name = "model_name_team-1_c0ffee"
patched_models.get_model_list = MagicMock(
return_value=[
{
"model_name": internal_name,
"model_info": {
"team_id": "team-1",
"team_public_model_name": "gpt-4-team",
},
}
]
)
patched_models.get_model_names = MagicMock(return_value=[internal_name])
async def _fake_get_available_models_for_user(**kwargs):
return [internal_name]
monkeypatch.setattr(
proxy_utils,
"get_available_models_for_user",
_fake_get_available_models_for_user,
)
monkeypatch.setattr(
model_checks, "get_complete_model_list", lambda **kwargs: [internal_name]
)
with auth_as():
response = client.get(
"/v1/models", params=params, headers={"anthropic-version": "2023-06-01"}
)
assert response.status_code == 200
assert [m["id"] for m in response.json()["data"]] == ["gpt-4-team"]
assert internal_name not in response.text
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
@pytest.mark.parametrize(
"caller_headers",
[
{"anthropic-version": "2023-06-01", "user-agent": "claude-code/2.1.267"},
{"anthropic-version": "2023-06-01", "user-agent": "claude-cli/2.1.267 (external, sdk-cli)"},
{"anthropic-version": "2023-06-01", "x-gateway-client": "claude-code"},
],
)
def test_anthropic_format_lists_claude_code_view_ids_for_claude_code(
client, auth_as, patched_models, monkeypatch, path, caller_headers
):
"""Claude Code drops every id without claude/anthropic in it and reads [1m] as its 1M marker, so for Claude
Code (its discovery fetch's own user agent, its SDK's, or the gateway-client header a launcher sends) every
group is listed under a Claude-shaped id with the marker where the window reaches 1M; the display name stays
the served name."""
def _create_model_info_response(model_id, provider="openai", **kwargs):
if model_id != "claude-sonnet":
return _stub_model_info_response(model_id=model_id, provider=provider)
return {**_stub_model_info_response(model_id=model_id, provider=provider), "max_input_tokens": 1000000}
patched_models.model_group_alias = {}
patched_models.has_model_id.return_value = False
patched_models.get_candidate_model_ids_for_route.side_effect = lambda name, team_id=None: frozenset({name}) if name in ("gpt-4", "claude-sonnet") else frozenset()
monkeypatch.setattr(proxy_utils, "create_model_info_response", _create_model_info_response)
with auth_as():
response = client.get(path, headers=caller_headers)
assert response.status_code == 200
body = response.json()
assert [(m["id"], m["display_name"]) for m in body["data"]] == [
("claude-router-6770742d34", "gpt-4"),
("claude-sonnet[1m]", "claude-sonnet"),
]
assert (body["first_id"], body["last_id"]) == ("claude-router-6770742d34", "claude-sonnet[1m]")
assert [row["source_model"] for row in body["data"]] == ["gpt-4", "claude-sonnet"]
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_anthropic_format_keeps_served_ids_for_other_anthropic_clients(client, auth_as, patched_models, path):
"""An Anthropic SDK asking for the vendor shape gets the served ids: the view is Claude Code's alone."""
with auth_as():
response = client.get(path, headers={"anthropic-version": "2023-06-01", "user-agent": "anthropic-sdk-python/0.40"})
assert response.status_code == 200
assert [m["id"] for m in response.json()["data"]] == ["gpt-4", "claude-sonnet"]
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
@pytest.mark.parametrize("params", [{}, {"scope": "expand"}])
def test_codex_format_when_client_version_present(client, auth_as, patched_models, path, params):
"""Codex CLI fetches a provider's catalog as ``GET /v1/models?client_version=<its version>`` and
decodes Codex's own ``{"models": [...]}`` shape; the same request without the parameter keeps the
OpenAI shape byte for byte."""
with auth_as():
codex_response = client.get(path, params={**params, "client_version": "0.159.3"})
openai_response = client.get(path, params=params)
assert codex_response.status_code == 200
assert codex_response.headers["content-type"] == "application/json"
body = codex_response.json()
assert list(body) == ["models"]
assert [(m["slug"], m["display_name"], m["priority"]) for m in body["models"]] == [
("gpt-4", "gpt-4", 0),
("claude-sonnet", "claude-sonnet", 1),
]
assert all(m["base_instructions"] and m["visibility"] == "list" for m in body["models"])
assert openai_response.status_code == 200
assert normalize(openai_response.json()) == {
"data": [
{"id": "<VOLATILE>", "object": "model", "created": "<VOLATILE>", "owned_by": "openai"},
{"id": "<VOLATILE>", "object": "model", "created": "<VOLATILE>", "owned_by": "openai"},
],
"object": "list",
}
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_codex_format_wins_over_the_anthropic_header(client, auth_as, patched_models, path):
with auth_as():
response = client.get(path, params={"client_version": "0.159.3"}, headers={"anthropic-version": "2023-06-01"})
assert response.status_code == 200
assert list(response.json()) == ["models"]
@pytest.mark.parametrize("path", ["/v1/models", "/models"])
def test_codex_format_carries_configured_service_tiers(client, auth_as, patched_models, path):
"""A deployment's ``model_info.service_tiers`` becomes the entry's ``service_tiers``, which Codex
offers as slash commands; a model without one offers none, and the OpenAI shape gains no field."""
patched_models.get_configured_service_tiers = MagicMock(
side_effect=lambda model_name, team_id=None: (["ultrafast"],) if model_name == "gpt-4" else (None,)
)
with auth_as():
codex_response = client.get(path, params={"client_version": "0.159.3"})
openai_response = client.get(path)
gpt_4, claude = codex_response.json()["models"]
assert gpt_4["service_tiers"] == [
{"id": "ultrafast", "name": "Ultrafast", "description": "Sends service_tier=ultrafast upstream"}
]
assert claude["service_tiers"] == []
assert all("service_tiers" not in m for m in openai_response.json()["data"])
@pytest.mark.parametrize("params", [{}, {"scope": "expand"}])
def test_codex_service_tiers_are_read_for_the_key_team(client, auth_as, patched_models, params):
"""A tier and the upstream model that picks Codex's stock entry are read off the deployments the
key's team can route to, so both listing paths hand the router the key's team, and no team for a
key without one."""
patched_models.get_configured_service_tiers = MagicMock(
side_effect=lambda model_name, team_id=None: (["ultrafast"],) if team_id == "team-1" else (None,)
)
patched_models.get_routable_upstream_model = MagicMock(
side_effect=lambda model_name, team_id=None: "openai/gpt-5.5" if team_id == "team-1" else "gpt-4"
)
with auth_as(team_id="team-1"):
team_response = client.get("/v1/models", params={**params, "client_version": "0.159.3"})
with auth_as():
teamless_response = client.get("/v1/models", params={**params, "client_version": "0.159.3"})
assert [[t["id"] for t in m["service_tiers"]] for m in team_response.json()["models"]] == [["ultrafast"]] * 2
assert [m["service_tiers"] for m in teamless_response.json()["models"]] == [[], []]
assert all(m["supported_reasoning_levels"] for m in team_response.json()["models"])
assert [m["supported_reasoning_levels"] for m in teamless_response.json()["models"]] == [[], []]
@pytest.mark.parametrize("params", [{}, {"scope": "expand"}])
def test_codex_service_tiers_resolved_via_internal_team_key(client, auth_as, patched_models, monkeypatch, params):
"""A team-scoped row's tiers are looked up by the internal routing key while the entry is keyed by
the public name Codex sends back as the model."""
from litellm.proxy import utils as proxy_utils
from litellm.proxy.auth import model_checks
internal_name = "model_name_team-1_c0ffee"
patched_models.get_model_list = MagicMock(
return_value=[
{"model_name": internal_name, "model_info": {"team_id": "team-1", "team_public_model_name": "gpt-4-team"}}
]
)
patched_models.get_model_names = MagicMock(return_value=[internal_name])
patched_models.get_configured_service_tiers = MagicMock(
side_effect=lambda model_name, team_id=None: (["ultrafast"],) if model_name == internal_name else ()
)
async def _fake_get_available_models_for_user(**kwargs):
return [internal_name]
monkeypatch.setattr(proxy_utils, "get_available_models_for_user", _fake_get_available_models_for_user)
monkeypatch.setattr(model_checks, "get_complete_model_list", lambda **kwargs: [internal_name])
with auth_as():
response = client.get("/v1/models", params={**params, "client_version": "0.159.3"})
assert response.status_code == 200
(entry,) = response.json()["models"]
assert (entry["slug"], [tier["id"] for tier in entry["service_tiers"]]) == ("gpt-4-team", ["ultrafast"])