From 3da6941dc7a5834615e84ccefed2f7173d0e76b5 Mon Sep 17 00:00:00 2001 From: ajitsharmas2007 Date: Sat, 26 Sep 2026 19:54:09 +0530 Subject: [PATCH 1/3] feat(proxy): list catalog-only models from general_settings.advertised_models Let an operator advertise a model id on GET /v1/models without registering a routable deployment for it, for models the proxy does not serve itself such as a realtime endpoint clients connect to directly. Previously the only way to make a name appear in the listing was to register a deployment, leaving a live route that failed confusingly when someone called it. Entries are discovery only: they register no route, so a request naming one still fails as an unknown model and GET /v1/models/{id} reports it as not found. An id already in the listing wins, so a catalog entry cannot displace or impersonate a routed model, and a repeated id is listed once. A listing asked for access groups alone carries none of them. Rows carry exactly the declared id and owner, never enriched from the cost map. A malformed setting is logged and skipped, leaving the listing unchanged. Fixes #40577 --- litellm/proxy/_types.py | 17 +++ .../proxy/common_utils/advertised_models.py | 88 ++++++++++++ litellm/proxy/proxy_server.py | 18 +++ .../common_utils/test_advertised_models.py | 93 ++++++++++++ .../test_model_list_advertised_models.py | 132 ++++++++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 39 ++++++ 6 files changed, 387 insertions(+) create mode 100644 litellm/proxy/common_utils/advertised_models.py create mode 100644 tests/test_litellm/proxy/common_utils/test_advertised_models.py create mode 100644 tests/test_litellm/proxy/test_model_list_advertised_models.py diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 3da30070b9e..34023245ee8 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2551,6 +2551,18 @@ class PluginConfig(LiteLLMPydanticObjectBase): ) +class AdvertisedModel(LiteLLMPydanticObjectBase): + """A catalog-only entry listed by /v1/models without a routable deployment. + + Discovery only: nothing here registers a route, so a request naming this id + still fails as an unknown model. Use it to advertise something the proxy + does not serve itself, e.g. a realtime endpoint clients connect to directly. + """ + + id: str = Field(description="model id as clients see it in the listing") + owned_by: str = Field(description="owner reported for this entry, e.g. the provider name") + + class CoordinationRedisNode(LiteLLMPydanticObjectBase): """A single startup node of a cluster-mode Redis used for proxy coordination.""" @@ -2644,6 +2656,11 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): plugins: list[PluginConfig] | None = Field( None, description="external services registered as embeddable UI plugins" ) + advertised_models: list[AdvertisedModel] | None = Field( + None, + description="catalog-only entries added to /v1/models for discovery; they register no route, so " + "GET /v1/models/{id} still reports them as not found and a request naming one fails as an unknown model", + ) key_management_system: KeyManagementSystem | None = Field( None, description="key manager to load keys from / decrypt keys with" ) diff --git a/litellm/proxy/common_utils/advertised_models.py b/litellm/proxy/common_utils/advertised_models.py new file mode 100644 index 00000000000..295fa1da09d --- /dev/null +++ b/litellm/proxy/common_utils/advertised_models.py @@ -0,0 +1,88 @@ +"""Catalog-only entries appended to the model listing. + +`general_settings.advertised_models` lets an operator advertise a model id on +`GET /v1/models` without registering a routable deployment for it. The use case +is a model clients reach on their own, e.g. a realtime endpoint over a +websocket: it belongs in the client's model picker, but the proxy never serves +it, and inventing a placeholder deployment just to make it appear would leave a +live route that fails confusingly when someone calls it. + +Discovery only, in both directions. Nothing here registers a route, so a request +naming a catalog id still fails as an unknown model, and `GET /v1/models/{id}` +still answers 404: the catalog says what exists, not what this proxy serves. + +A catalog entry never displaces a routed one. An id already in the listing wins, +so a typo here cannot mask a working model or impersonate it, and a repeated id +is listed once. + +A row carries exactly what the operator declared. Nothing is inferred from the +cost map, so an entry that happens to share a name with a known model does not +silently pick up that model's mode or token limits, and an entry the cost map +cannot resolve does not drive `get_llm_provider` into printing its provider list +to stdout on every listing request. + +Misconfiguration fails open: an `advertised_models` value that does not validate +is logged and skipped, leaving the listing exactly as it would have been. +""" + +from __future__ import annotations + +from collections.abc import Collection, Mapping +from typing import Final + +from pydantic import TypeAdapter, ValidationError + +from litellm._logging import verbose_proxy_logger +from litellm.constants import DEFAULT_MODEL_CREATED_AT_TIME +from litellm.proxy._types import AdvertisedModel +from litellm.types.proxy.model_listing import ModelInfoResponse + +ADVERTISED_MODELS_SETTING: Final = "advertised_models" + +_ADVERTISED_MODELS_ADAPTER: Final = TypeAdapter(tuple[AdvertisedModel, ...]) + + +def configured_advertised_models(general_settings: Mapping[str, object]) -> tuple[AdvertisedModel, ...]: + """Typed `advertised_models` entries, or none when unset or malformed.""" + configured: Final = general_settings.get(ADVERTISED_MODELS_SETTING) + if configured is None: + return () + try: + return _ADVERTISED_MODELS_ADAPTER.validate_python(configured) + except ValidationError as validation_error: + verbose_proxy_logger.warning( + "general_settings.%s is not a list of {id, owned_by} entries, skipping it: %s", + ADVERTISED_MODELS_SETTING, + validation_error, + ) + return () + + +def _listing_row(entry: AdvertisedModel) -> ModelInfoResponse: + row: Final[ModelInfoResponse] = { + "id": entry.id, + "object": "model", + "created": DEFAULT_MODEL_CREATED_AT_TIME, + "owned_by": entry.owned_by, + } + return row + + +def advertised_model_rows( + general_settings: Mapping[str, object], + listed_ids: Collection[str], +) -> tuple[ModelInfoResponse, ...]: + """Listing rows for the configured catalog entries, in configured order. + + `listed_ids` are the ids the listing already carries; entries naming one of + them are dropped so a routed model is never displaced. + """ + already_listed: Final = frozenset(listed_ids) + candidates: Final = tuple( + entry for entry in configured_advertised_models(general_settings) if entry.id not in already_listed + ) + return tuple( + _listing_row(entry) + for index, entry in enumerate(candidates) + if all(earlier.id != entry.id for earlier in candidates[:index]) + ) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index ed4ea347c2e..c6e2348726a 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -394,6 +394,7 @@ from litellm.proxy.common_request_processing import ( resolve_litellm_call_id, ttft_keepalive_interval, ) +from litellm.proxy.common_utils.advertised_models import advertised_model_rows from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import ( AuthCacheInvalidationSubscriber, ) @@ -11260,6 +11261,13 @@ async def model_list( configured (cooldown remains the sole exclusion mechanism). Hiding is presentation-only: a hidden model can still be called directly. + + Set `general_settings.advertised_models` to add catalog-only entries, each + an `{id, owned_by}` pair, for models this proxy does not serve itself (say a + realtime endpoint clients connect to directly). They are listed for + discovery only and register no route, so a request naming one fails as an + unknown model and `GET /v1/models/{id}` reports it as not found. An entry + whose id is already listed is dropped, so a routed model is never displaced. """ global llm_model_list, general_settings, llm_router, prisma_client, user_api_key_cache, proxy_logging_obj @@ -11383,6 +11391,12 @@ async def model_list( model_info["id"] = response_id model_data.append(model_info) + # Catalog-only entries are advertised to every caller: they name no deployment, + # so the per-caller filters above have nothing to scope them by. A listing asked + # for access groups alone gets none of them, since a catalog entry is not a group. + if not only_model_access_groups: + model_data.extend(advertised_model_rows(settings, tuple(row["id"] for row in model_data))) + if wants_anthropic_format: admin_listing: Final = cast(Sequence[ModelInfoResponse], model_data) # cast-ok: rows built above return create_anthropic_model_list_response( @@ -11443,6 +11457,10 @@ async def model_list( model_info["id"] = response_id model_data.append(model_info) + # Same catalog merge as the scope=expand branch above. + if not only_model_access_groups: + model_data.extend(advertised_model_rows(settings, tuple(row["id"] for row in model_data))) + if wants_anthropic_format: listing: Final = cast(Sequence[ModelInfoResponse], model_data) # cast-ok: rows built above return create_anthropic_model_list_response( diff --git a/tests/test_litellm/proxy/common_utils/test_advertised_models.py b/tests/test_litellm/proxy/common_utils/test_advertised_models.py new file mode 100644 index 00000000000..7495432928f --- /dev/null +++ b/tests/test_litellm/proxy/common_utils/test_advertised_models.py @@ -0,0 +1,93 @@ +"""Tests for `general_settings.advertised_models`, the catalog-only entries +appended to the model listing by `litellm.proxy.common_utils.advertised_models`. +""" + +from typing import Final + +import litellm +from litellm.constants import DEFAULT_MODEL_CREATED_AT_TIME +from litellm.proxy.common_utils.advertised_models import ( + advertised_model_rows, + configured_advertised_models, +) + +CATALOG_ENTRY: Final = {"id": "catalog-only-model", "owned_by": "example-provider"} + + +def _settings(*entries: object) -> dict[str, object]: + return {"advertised_models": list(entries)} + + +def test_entry_becomes_a_row_carrying_its_configured_id_and_owner(): + rows: Final = advertised_model_rows(_settings(CATALOG_ENTRY), []) + + assert rows == ( + { + "id": "catalog-only-model", + "object": "model", + "created": DEFAULT_MODEL_CREATED_AT_TIME, + "owned_by": "example-provider", + }, + ), f"expected one row carrying exactly the configured id and owner, got {rows}" + + +def test_entry_naming_an_already_listed_model_is_dropped(): + """A routed model is never displaced by a catalog entry that shares its id.""" + rows: Final = advertised_model_rows(_settings({"id": "routed-model", "owned_by": "impostor"}), ["routed-model"]) + + assert rows == (), f"catalog entry must not shadow the routed model already listed, got {rows}" + + +def test_repeated_id_is_listed_once_keeping_the_first_entry(): + rows: Final = advertised_model_rows( + _settings(CATALOG_ENTRY, {"id": "catalog-only-model", "owned_by": "second-declaration"}), + [], + ) + + assert [row["owned_by"] for row in rows] == ["example-provider"], ( + f"a repeated id should be listed once, keeping the first declaration, got {rows}" + ) + + +def test_no_setting_produces_no_rows(): + assert advertised_model_rows({}, []) == (), "an unset advertised_models must add nothing to the listing" + + +def test_malformed_setting_is_skipped_rather_than_raising(): + """Misconfiguration fails open: the listing is returned as if nothing was set.""" + missing_owner: Final = advertised_model_rows({"advertised_models": [{"id": "no-owner"}]}, []) + not_a_list: Final = advertised_model_rows({"advertised_models": "catalog-only-model"}, []) + + assert missing_owner == (), f"an entry without owned_by must be skipped, got {missing_owner}" + assert not_a_list == (), f"a non-list advertised_models must be skipped, got {not_a_list}" + + +def test_row_carries_only_declared_fields_even_for_a_model_the_cost_map_knows(): + """Catalog rows are exactly what the operator declared, never enriched. + + The id is taken from the live cost map rather than hard-coded, so this keeps + testing the no-enrichment guarantee as the catalog changes. + """ + known_model: Final = next(iter(litellm.model_cost)) + + rows: Final = advertised_model_rows(_settings({"id": known_model, "owned_by": "example-provider"}), []) + + assert rows == ( + { + "id": known_model, + "object": "model", + "created": DEFAULT_MODEL_CREATED_AT_TIME, + "owned_by": "example-provider", + }, + ), f"a catalog row must not pick up cost map details for {known_model}, got {rows}" + + +def test_configured_entries_are_typed_and_keep_their_order(): + entries: Final = configured_advertised_models( + _settings(CATALOG_ENTRY, {"id": "second", "owned_by": "other-provider"}) + ) + + assert [(entry.id, entry.owned_by) for entry in entries] == [ + ("catalog-only-model", "example-provider"), + ("second", "other-provider"), + ], f"entries should be parsed in configured order, got {entries}" diff --git a/tests/test_litellm/proxy/test_model_list_advertised_models.py b/tests/test_litellm/proxy/test_model_list_advertised_models.py new file mode 100644 index 00000000000..6bcf661b793 --- /dev/null +++ b/tests/test_litellm/proxy/test_model_list_advertised_models.py @@ -0,0 +1,132 @@ +"""Tests for catalog-only entries on GET /v1/models. + +`general_settings.advertised_models` adds entries to the listing for models the +proxy does not serve itself. They are discovery only: no deployment is +registered, so routing is untouched and a routed model is never displaced. +""" + +from typing import Final +from unittest.mock import AsyncMock, MagicMock + +import pytest + +from litellm.proxy import proxy_server +from litellm.proxy._types import UserAPIKeyAuth + +CATALOG_SETTINGS: Final = {"advertised_models": [{"id": "catalog-only-model", "owned_by": "example-provider"}]} + + +@pytest.fixture +def patched_model_list(monkeypatch): + """Stub router + utility helpers used by `model_list`.""" + from litellm.proxy import utils as proxy_utils + + router: Final = MagicMock() + router.get_fully_blocked_model_names = MagicMock(return_value=set()) + router.async_get_fully_unhealthy_model_names = AsyncMock(return_value=set()) + router.get_model_names = MagicMock(return_value=["routed-model"]) + router.get_model_access_groups = MagicMock(return_value={}) + + monkeypatch.setattr(proxy_server, "llm_router", router) + monkeypatch.setattr(proxy_server, "user_model", None) + monkeypatch.setattr(proxy_server, "general_settings", {}) + + async def _fake_get_available_models_for_user(**kwargs): + return ["routed-model"] + + monkeypatch.setattr(proxy_utils, "get_available_models_for_user", _fake_get_available_models_for_user) + + def _fake_create_model_info_response(model_id, provider="openai", **kwargs): + return {"id": model_id, "object": "model", "created": 0, "owned_by": provider} + + monkeypatch.setattr(proxy_utils, "create_model_info_response", _fake_create_model_info_response) + + return router + + +async def _listing(**kwargs): + response: Final = await proxy_server.model_list(user_api_key_dict=UserAPIKeyAuth(api_key="sk-test"), **kwargs) + return response["data"] + + +@pytest.mark.asyncio +async def test_catalog_entry_is_listed_alongside_routed_models(patched_model_list, monkeypatch): + monkeypatch.setattr(proxy_server, "general_settings", CATALOG_SETTINGS) + + rows: Final = await _listing() + + assert [row["id"] for row in rows] == [ + "routed-model", + "catalog-only-model", + ], f"the catalog entry should be listed after the routed models, got {rows}" + assert rows[1]["owned_by"] == "example-provider", f"the configured owner should be reported, got {rows[1]}" + + +@pytest.mark.asyncio +async def test_routed_rows_are_untouched_by_the_catalog(patched_model_list, monkeypatch): + """The routed model's row is identical with and without catalog entries.""" + without_catalog: Final = await _listing() + + monkeypatch.setattr(proxy_server, "general_settings", CATALOG_SETTINGS) + with_catalog: Final = await _listing() + + assert with_catalog[0] == without_catalog[0], ( + "configuring advertised_models must not change an existing row: " + f"{with_catalog[0]} differs from {without_catalog[0]}" + ) + + +@pytest.mark.asyncio +async def test_listing_is_unchanged_when_no_catalog_is_configured(patched_model_list): + rows: Final = await _listing() + + assert rows == [{"id": "routed-model", "object": "model", "created": 0, "owned_by": "openai"}], ( + f"with no advertised_models the listing must be exactly what it was before, got {rows}" + ) + + +@pytest.mark.asyncio +async def test_catalog_entry_cannot_displace_a_routed_model_of_the_same_id(patched_model_list, monkeypatch): + monkeypatch.setattr( + proxy_server, + "general_settings", + {"advertised_models": [{"id": "routed-model", "owned_by": "impostor"}]}, + ) + + rows: Final = await _listing() + + assert rows == [{"id": "routed-model", "object": "model", "created": 0, "owned_by": "openai"}], ( + f"the routed model must be listed once, unchanged, got {rows}" + ) + + +@pytest.mark.asyncio +async def test_access_group_only_listing_carries_no_catalog_entries(patched_model_list, monkeypatch): + """A caller asking for access groups alone gets none: a catalog entry is not a group.""" + monkeypatch.setattr(proxy_server, "general_settings", CATALOG_SETTINGS) + + rows: Final = await _listing(only_model_access_groups=True) + + assert "catalog-only-model" not in [row["id"] for row in rows], ( + f"only_model_access_groups must not surface catalog entries, got {rows}" + ) + + +@pytest.mark.asyncio +async def test_catalog_entry_is_listed_for_scope_expand(patched_model_list, monkeypatch): + from litellm.proxy.auth import model_checks + from litellm.proxy.management_endpoints import common_utils + + async def _fake_admin(**kwargs): + return True + + monkeypatch.setattr(common_utils, "_user_has_admin_privileges", _fake_admin) + monkeypatch.setattr(model_checks, "get_complete_model_list", lambda **kwargs: ["routed-model"]) + monkeypatch.setattr(proxy_server, "general_settings", CATALOG_SETTINGS) + + rows: Final = await _listing(scope="expand") + + assert [row["id"] for row in rows] == [ + "routed-model", + "catalog-only-model", + ], f"the admin listing should carry catalog entries too, got {rows}" diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index bba67bcf6c2..af810a1b0a5 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -9798,6 +9798,13 @@ export interface paths { * configured (cooldown remains the sole exclusion mechanism). * Hiding is presentation-only: a hidden model can still be * called directly. + * + * Set `general_settings.advertised_models` to add catalog-only entries, each + * an `{id, owned_by}` pair, for models this proxy does not serve itself (say a + * realtime endpoint clients connect to directly). They are listed for + * discovery only and register no route, so a request naming one fails as an + * unknown model and `GET /v1/models/{id}` reports it as not found. An entry + * whose id is already listed is dropped, so a routed model is never displaced. */ get: operations["model_list_models_get"]; put?: never; @@ -20104,6 +20111,13 @@ export interface paths { * configured (cooldown remains the sole exclusion mechanism). * Hiding is presentation-only: a hidden model can still be * called directly. + * + * Set `general_settings.advertised_models` to add catalog-only entries, each + * an `{id, owned_by}` pair, for models this proxy does not serve itself (say a + * realtime endpoint clients connect to directly). They are listed for + * discovery only and register no route, so a request naming one fails as an + * unknown model and `GET /v1/models/{id}` reports it as not found. An entry + * whose id is already listed is dropped, so a routed model is never displaced. */ get: operations["model_list_v1_models_get"]; put?: never; @@ -24128,6 +24142,26 @@ export interface components { [key: string]: string; }; }; + /** + * AdvertisedModel + * @description A catalog-only entry listed by /v1/models without a routable deployment. + * + * Discovery only: nothing here registers a route, so a request naming this id + * still fails as an unknown model. Use it to advertise something the proxy + * does not serve itself, e.g. a realtime endpoint clients connect to directly. + */ + AdvertisedModel: { + /** + * Id + * @description model id as clients see it in the listing + */ + id: string; + /** + * Owned By + * @description owner reported for this entry, e.g. the provider name + */ + owned_by: string; + }; /** * AgentCapabilities * @description Defines optional capabilities supported by an agent. @@ -28013,6 +28047,11 @@ export interface components { * @default 1 */ admission_queue_timeout_seconds: number; + /** + * Advertised Models + * @description catalog-only entries added to /v1/models for discovery; they register no route, so GET /v1/models/{id} still reports them as not found and a request naming one fails as an unknown model + */ + advertised_models?: components["schemas"]["AdvertisedModel"][] | null; /** * Alert To Webhook Url * @description Mapping of alert type to webhook url. e.g. `alert_to_webhook_url: {'budget_alerts': 'https://nothooks.slack.com/services/T00000000/B00000000/XXXXXXXXXXXXXXXXXXXXXXXX'}` From 3eaad4fb822f4638650a70ac68028fa3a305da0c Mon Sep 17 00:00:00 2001 From: ajitsharmas2007 Date: Sat, 26 Sep 2026 20:55:59 +0530 Subject: [PATCH 2/3] fix(proxy): keep advertised models inside the caller's listing authorization Review of #43310 surfaced two ways a catalog entry escaped the filtering the routed rows go through, plus two smaller issues. Reserve every name the router knows, not just the ids left in this caller's listing. The collision check compared against the post-filter listing, so a deployment removed by team scoping, a pause or the health filter was treated as a free id: an entry naming it put it back, under whatever owner the config declared. A key restricted to one model could be shown two deployments it had no access to. Access group names are reserved too, so an entry cannot shadow a group. Run catalog ids through the same listing callbacks as routed rows, so a callback that hides one is honored instead of bypassed. Mark every catalog row `catalog_only`, since /models feeds the team and key model pickers and an unroutable id was indistinguishable from a routable one there. Deduplicate in one pass rather than rescanning earlier entries, and drop two comments that restated the code beside them. --- .../proxy/common_utils/advertised_models.py | 59 +++++++++++----- litellm/proxy/proxy_server.py | 25 +++++-- litellm/types/proxy/model_listing.py | 7 +- .../common_utils/test_advertised_models.py | 63 +++++++++++++++++ .../test_model_list_advertised_models.py | 69 +++++++++++++++++++ 5 files changed, 198 insertions(+), 25 deletions(-) diff --git a/litellm/proxy/common_utils/advertised_models.py b/litellm/proxy/common_utils/advertised_models.py index 295fa1da09d..d7c14bda446 100644 --- a/litellm/proxy/common_utils/advertised_models.py +++ b/litellm/proxy/common_utils/advertised_models.py @@ -11,9 +11,14 @@ Discovery only, in both directions. Nothing here registers a route, so a request naming a catalog id still fails as an unknown model, and `GET /v1/models/{id}` still answers 404: the catalog says what exists, not what this proxy serves. -A catalog entry never displaces a routed one. An id already in the listing wins, -so a typo here cannot mask a working model or impersonate it, and a repeated id -is listed once. +A catalog entry never displaces a routed one, and never resurrects one. The +reserved set is every name the router knows plus every id already listed, not +just what this caller can see, so an entry cannot re-expose a model that team +scoping, a pause or a health filter had hidden from them, nor impersonate it +under a different owner. A repeated id is listed once. + +Every row is marked `catalog_only`, so a client picking models out of the +listing can tell an advertised id from one the proxy will actually route. A row carries exactly what the operator declared. Nothing is inferred from the cost map, so an entry that happens to share a name with a known model does not @@ -28,7 +33,8 @@ is logged and skipped, leaving the listing exactly as it would have been. from __future__ import annotations from collections.abc import Collection, Mapping -from typing import Final +from types import MappingProxyType +from typing import TYPE_CHECKING, Final from pydantic import TypeAdapter, ValidationError @@ -37,6 +43,9 @@ from litellm.constants import DEFAULT_MODEL_CREATED_AT_TIME from litellm.proxy._types import AdvertisedModel from litellm.types.proxy.model_listing import ModelInfoResponse +if TYPE_CHECKING: + from litellm.router import Router + ADVERTISED_MODELS_SETTING: Final = "advertised_models" _ADVERTISED_MODELS_ADAPTER: Final = TypeAdapter(tuple[AdvertisedModel, ...]) @@ -64,25 +73,39 @@ def _listing_row(entry: AdvertisedModel) -> ModelInfoResponse: "object": "model", "created": DEFAULT_MODEL_CREATED_AT_TIME, "owned_by": entry.owned_by, + "catalog_only": True, } return row +def _reserved_ids(listed_ids: Collection[str], llm_router: Router | None) -> frozenset[str]: + """Ids a catalog entry may not claim. + + Every name the router knows counts, not just the ids this caller can see, so + an entry cannot re-expose a deployment that team scoping, a pause or a health + filter had already removed from their listing. + """ + if llm_router is None: + return frozenset(listed_ids) + return frozenset(listed_ids).union(llm_router.get_model_names(), llm_router.get_model_access_groups()) + + +def _first_per_id(entries: tuple[AdvertisedModel, ...]) -> tuple[AdvertisedModel, ...]: + """The first entry declared for each id, in configured order.""" + latest_wins: Final = MappingProxyType({entry.id: entry for entry in reversed(entries)}) + ordered_ids: Final = tuple(dict.fromkeys(entry.id for entry in entries)) + return tuple(latest_wins[entry_id] for entry_id in ordered_ids) + + def advertised_model_rows( general_settings: Mapping[str, object], listed_ids: Collection[str], + llm_router: Router | None = None, ) -> tuple[ModelInfoResponse, ...]: - """Listing rows for the configured catalog entries, in configured order. - - `listed_ids` are the ids the listing already carries; entries naming one of - them are dropped so a routed model is never displaced. - """ - already_listed: Final = frozenset(listed_ids) - candidates: Final = tuple( - entry for entry in configured_advertised_models(general_settings) if entry.id not in already_listed - ) - return tuple( - _listing_row(entry) - for index, entry in enumerate(candidates) - if all(earlier.id != entry.id for earlier in candidates[:index]) - ) + """Listing rows for the configured catalog entries, in configured order.""" + configured: Final = configured_advertised_models(general_settings) + if not configured: + return () + reserved: Final = _reserved_ids(listed_ids, llm_router) + candidates: Final = tuple(entry for entry in configured if entry.id not in reserved) + return tuple(_listing_row(entry) for entry in _first_per_id(candidates)) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index c6e2348726a..1c340217f73 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -177,6 +177,7 @@ if TYPE_CHECKING: from litellm.integrations.opentelemetry import OpenTelemetry from litellm.proxy.health_check_utils.shared_health_check_manager import SharedHealthCheckManager + from litellm.types.proxy.model_listing import ModelInfoResponse Span = _Span | Any else: @@ -11213,6 +11214,20 @@ async def _entries_kept_by_listing_callbacks( return tuple(entry for entry in entries if entry[0] not in hidden) +async def _catalog_listing_rows( + settings: Mapping[str, object], + listed_ids: Sequence[str], + user_api_key_dict: UserAPIKeyAuth, +) -> tuple["ModelInfoResponse", ...]: + """Catalog-only rows for this caller, filtered by the same listing callbacks + the routed rows went through, so an operator's callback can hide them too.""" + rows: Final = advertised_model_rows(settings, listed_ids, llm_router) + if not rows: + return () + hidden: Final = await _names_hidden_by_listing_callbacks(user_api_key_dict, tuple(row["id"] for row in rows)) + return tuple(row for row in rows if row["id"] not in hidden) + + async def _deployment_hidden_by_listing_callbacks(deployment: Deployment, user_api_key_dict: UserAPIKeyAuth) -> bool: listed_name: Final = _translate_model_name_for_response(deployment.model_dump(exclude_none=True)).get("model_name") if not isinstance(listed_name, str): @@ -11391,11 +11406,10 @@ async def model_list( model_info["id"] = response_id model_data.append(model_info) - # Catalog-only entries are advertised to every caller: they name no deployment, - # so the per-caller filters above have nothing to scope them by. A listing asked - # for access groups alone gets none of them, since a catalog entry is not a group. if not only_model_access_groups: - model_data.extend(advertised_model_rows(settings, tuple(row["id"] for row in model_data))) + model_data.extend( + await _catalog_listing_rows(settings, [row["id"] for row in model_data], user_api_key_dict) + ) if wants_anthropic_format: admin_listing: Final = cast(Sequence[ModelInfoResponse], model_data) # cast-ok: rows built above @@ -11457,9 +11471,8 @@ async def model_list( model_info["id"] = response_id model_data.append(model_info) - # Same catalog merge as the scope=expand branch above. if not only_model_access_groups: - model_data.extend(advertised_model_rows(settings, tuple(row["id"] for row in model_data))) + model_data.extend(await _catalog_listing_rows(settings, [row["id"] for row in model_data], user_api_key_dict)) if wants_anthropic_format: listing: Final = cast(Sequence[ModelInfoResponse], model_data) # cast-ok: rows built above diff --git a/litellm/types/proxy/model_listing.py b/litellm/types/proxy/model_listing.py index 24cfa85eee4..0f001400186 100644 --- a/litellm/types/proxy/model_listing.py +++ b/litellm/types/proxy/model_listing.py @@ -2,7 +2,7 @@ from typing import Literal -from typing_extensions import NotRequired, TypedDict +from typing_extensions import NotRequired, ReadOnly, TypedDict class ModelInfoMetadata(TypedDict): @@ -13,6 +13,10 @@ class ModelInfoResponse(TypedDict): """OpenAI-compatible model object. `mode`, `max_input_tokens`, and `max_output_tokens` are attached when the cost map or deployment config knows them; `metadata` is present only with include_metadata=true. + + `catalog_only` marks a row the proxy advertises but does not route, so a + client choosing a model from this listing can skip the ones a request would + be rejected for. It is absent on every routable model. """ id: str @@ -23,3 +27,4 @@ class ModelInfoResponse(TypedDict): max_input_tokens: NotRequired[int] max_output_tokens: NotRequired[int] metadata: NotRequired[ModelInfoMetadata] + catalog_only: NotRequired[ReadOnly[Literal[True]]] diff --git a/tests/test_litellm/proxy/common_utils/test_advertised_models.py b/tests/test_litellm/proxy/common_utils/test_advertised_models.py index 7495432928f..88fe501d9b8 100644 --- a/tests/test_litellm/proxy/common_utils/test_advertised_models.py +++ b/tests/test_litellm/proxy/common_utils/test_advertised_models.py @@ -27,6 +27,7 @@ def test_entry_becomes_a_row_carrying_its_configured_id_and_owner(): "object": "model", "created": DEFAULT_MODEL_CREATED_AT_TIME, "owned_by": "example-provider", + "catalog_only": True, }, ), f"expected one row carrying exactly the configured id and owner, got {rows}" @@ -78,6 +79,7 @@ def test_row_carries_only_declared_fields_even_for_a_model_the_cost_map_knows(): "object": "model", "created": DEFAULT_MODEL_CREATED_AT_TIME, "owned_by": "example-provider", + "catalog_only": True, }, ), f"a catalog row must not pick up cost map details for {known_model}, got {rows}" @@ -91,3 +93,64 @@ def test_configured_entries_are_typed_and_keep_their_order(): ("catalog-only-model", "example-provider"), ("second", "other-provider"), ], f"entries should be parsed in configured order, got {entries}" + + +class _RouterStub: + """Minimal stand-in for the bits of Router this module reads.""" + + def __init__(self, names: tuple[str, ...], groups: tuple[str, ...] = ()) -> None: + self._names: Final = names + self._groups: Final = groups + + def get_model_names(self) -> list[str]: + return list(self._names) + + def get_model_access_groups(self) -> dict[str, list[str]]: + return {group: [] for group in self._groups} + + +def test_entry_naming_a_routed_model_hidden_from_this_caller_is_dropped(): + """A catalog entry must not re-expose a deployment the listing filtered out. + + `listed_ids` carries only what this caller may see, so an id that was scoped + away, paused or health-filtered would otherwise reappear under any owner the + config names. + """ + rows: Final = advertised_model_rows( + _settings({"id": "hidden-deployment", "owned_by": "impostor"}), + [], + _RouterStub(("hidden-deployment",)), + ) + + assert rows == (), f"a routed model absent from this caller's listing must stay absent, got {rows}" + + +def test_entry_naming_an_access_group_is_dropped(): + rows: Final = advertised_model_rows( + _settings({"id": "beta-models", "owned_by": "impostor"}), + [], + _RouterStub((), ("beta-models",)), + ) + + assert rows == (), f"a catalog entry must not shadow an access group name, got {rows}" + + +def test_entry_is_listed_when_the_router_knows_nothing_about_it(): + rows: Final = advertised_model_rows( + _settings(CATALOG_ENTRY), + ["routed-model"], + _RouterStub(("routed-model", "another-model")), + ) + + assert [row["id"] for row in rows] == ["catalog-only-model"], ( + f"an id no deployment claims should still be listed, got {rows}" + ) + + +def test_every_row_is_marked_catalog_only(): + rows: Final = advertised_model_rows(_settings(CATALOG_ENTRY, {"id": "second", "owned_by": "other"}), []) + + assert [row.get("catalog_only") for row in rows] == [ + True, + True, + ], f"every catalog row must be marked so clients can tell it from a routable model, got {rows}" diff --git a/tests/test_litellm/proxy/test_model_list_advertised_models.py b/tests/test_litellm/proxy/test_model_list_advertised_models.py index 6bcf661b793..f106c587146 100644 --- a/tests/test_litellm/proxy/test_model_list_advertised_models.py +++ b/tests/test_litellm/proxy/test_model_list_advertised_models.py @@ -10,8 +10,11 @@ from unittest.mock import AsyncMock, MagicMock import pytest +import litellm +from litellm.integrations.custom_logger import CustomLogger from litellm.proxy import proxy_server from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy.utils import ProxyLogging CATALOG_SETTINGS: Final = {"advertised_models": [{"id": "catalog-only-model", "owned_by": "example-provider"}]} @@ -130,3 +133,69 @@ async def test_catalog_entry_is_listed_for_scope_expand(patched_model_list, monk "routed-model", "catalog-only-model", ], f"the admin listing should carry catalog entries too, got {rows}" + + +@pytest.mark.asyncio +async def test_catalog_entry_cannot_resurrect_a_model_this_caller_may_not_see(patched_model_list, monkeypatch): + """A deployment filtered out of this caller's listing stays out. + + The router knows `restricted-model`, but this key cannot see it, so it never + reaches `model_data`. Advertising that id must not put it back. + """ + patched_model_list.get_model_names = MagicMock(return_value=["routed-model", "restricted-model"]) + monkeypatch.setattr( + proxy_server, + "general_settings", + {"advertised_models": [{"id": "restricted-model", "owned_by": "impostor"}]}, + ) + + rows: Final = await _listing() + + assert [row["id"] for row in rows] == ["routed-model"], ( + f"a deployment hidden from this caller must not reappear as a catalog entry, got {rows}" + ) + + +@pytest.mark.asyncio +async def test_catalog_rows_are_marked_and_routed_rows_are_not(patched_model_list, monkeypatch): + monkeypatch.setattr(proxy_server, "general_settings", CATALOG_SETTINGS) + + rows: Final = await _listing() + + assert [(row["id"], row.get("catalog_only")) for row in rows] == [ + ("routed-model", None), + ("catalog-only-model", True), + ], f"only catalog rows carry the marker, so a client can tell them apart, got {rows}" + + +class _HidingGate(CustomLogger): + """A listing callback that hides whichever names it was given.""" + + def __init__(self, hidden: frozenset[str]) -> None: + super().__init__() + self.hidden = hidden + self.seen: list[tuple[str, ...]] = [] + + async def async_filter_listed_models(self, user_api_key_dict, model_names): + self.seen.append(tuple(model_names)) + return [name for name in model_names if name not in self.hidden] + + +@pytest.mark.asyncio +async def test_listing_callbacks_can_hide_a_catalog_entry(patched_model_list, monkeypatch): + """Catalog rows go through the same per-caller callbacks as routed rows. + + Appending them afterwards would let an entry past a callback that was meant + to hide it. + """ + gate: Final = _HidingGate(frozenset({"catalog-only-model"})) + monkeypatch.setattr(litellm, "callbacks", [gate]) + ProxyLogging._callback_capabilities_cache.clear() + monkeypatch.setattr(proxy_server, "general_settings", CATALOG_SETTINGS) + + rows: Final = await _listing() + + assert [row["id"] for row in rows] == ["routed-model"], ( + f"a callback that hides a catalog id must keep it out of the listing, got {rows}" + ) + assert ("catalog-only-model",) in gate.seen, f"the catalog id must be offered to the callback, saw {gate.seen}" From b9d86129b5c8c56d6451e2ce1563cb02b004eb62 Mon Sep 17 00:00:00 2001 From: ajitsharmas2007 Date: Sat, 26 Sep 2026 22:22:54 +0530 Subject: [PATCH 3/3] fix(proxy): reserve team-public model names and group aliases from the catalog get_model_names() is not the set of routable names. Given no team id it drops team-scoped deployments along with their public name, and despite its docstring it never returns model_group_alias keys. An advertised entry could therefore claim either kind of name and disclose it to a caller who cannot list the model behind it. Read team_public_model_names and model_group_alias off the router instead, and record on the helper why get_model_names() alone will not do. Also record the test callback's observations in a tuple rather than growing a list. --- .../proxy/common_utils/advertised_models.py | 17 +++++++--- .../common_utils/test_advertised_models.py | 33 ++++++++++++++++++- .../test_model_list_advertised_models.py | 4 +-- 3 files changed, 47 insertions(+), 7 deletions(-) diff --git a/litellm/proxy/common_utils/advertised_models.py b/litellm/proxy/common_utils/advertised_models.py index d7c14bda446..815fdf76a8a 100644 --- a/litellm/proxy/common_utils/advertised_models.py +++ b/litellm/proxy/common_utils/advertised_models.py @@ -81,13 +81,22 @@ def _listing_row(entry: AdvertisedModel) -> ModelInfoResponse: def _reserved_ids(listed_ids: Collection[str], llm_router: Router | None) -> frozenset[str]: """Ids a catalog entry may not claim. - Every name the router knows counts, not just the ids this caller can see, so - an entry cannot re-expose a deployment that team scoping, a pause or a health - filter had already removed from their listing. + Every routable name counts, not just the ids this caller can see, so an entry + cannot re-expose a deployment that team scoping, a pause or a health filter + had already removed from their listing. + + `get_model_names()` alone is not that set: with no team id it drops + team-scoped deployments, and despite its docstring it never returns + `model_group_alias` keys, so both are unioned in explicitly. """ if llm_router is None: return frozenset(listed_ids) - return frozenset(listed_ids).union(llm_router.get_model_names(), llm_router.get_model_access_groups()) + return frozenset(listed_ids).union( + llm_router.get_model_names(), + llm_router.get_model_access_groups(), + llm_router.team_public_model_names, + llm_router.model_group_alias or (), + ) def _first_per_id(entries: tuple[AdvertisedModel, ...]) -> tuple[AdvertisedModel, ...]: diff --git a/tests/test_litellm/proxy/common_utils/test_advertised_models.py b/tests/test_litellm/proxy/common_utils/test_advertised_models.py index 88fe501d9b8..eda8e98e20d 100644 --- a/tests/test_litellm/proxy/common_utils/test_advertised_models.py +++ b/tests/test_litellm/proxy/common_utils/test_advertised_models.py @@ -98,11 +98,20 @@ def test_configured_entries_are_typed_and_keep_their_order(): class _RouterStub: """Minimal stand-in for the bits of Router this module reads.""" - def __init__(self, names: tuple[str, ...], groups: tuple[str, ...] = ()) -> None: + def __init__( + self, + names: tuple[str, ...], + groups: tuple[str, ...] = (), + team_public: frozenset[str] = frozenset(), + aliases: tuple[str, ...] = (), + ) -> None: self._names: Final = names self._groups: Final = groups + self.team_public_model_names: Final = team_public + self.model_group_alias: Final = {alias: "some-target" for alias in aliases} def get_model_names(self) -> list[str]: + """Mirrors the real method: team-scoped names and aliases are absent here.""" return list(self._names) def get_model_access_groups(self) -> dict[str, list[str]]: @@ -154,3 +163,25 @@ def test_every_row_is_marked_catalog_only(): True, True, ], f"every catalog row must be marked so clients can tell it from a routable model, got {rows}" + + +def test_entry_naming_a_team_scoped_public_name_is_dropped(): + """`get_model_names()` omits team-scoped deployments when given no team id.""" + rows: Final = advertised_model_rows( + _settings({"id": "team-public-gpt", "owned_by": "impostor"}), + [], + _RouterStub((), team_public=frozenset({"team-public-gpt"})), + ) + + assert rows == (), f"a team's public model name must stay reserved, got {rows}" + + +def test_entry_naming_a_model_group_alias_is_dropped(): + """`get_model_names()` does not return alias keys despite its docstring.""" + rows: Final = advertised_model_rows( + _settings({"id": "gpt-alias", "owned_by": "impostor"}), + [], + _RouterStub((), aliases=("gpt-alias",)), + ) + + assert rows == (), f"a routable alias must stay reserved, got {rows}" diff --git a/tests/test_litellm/proxy/test_model_list_advertised_models.py b/tests/test_litellm/proxy/test_model_list_advertised_models.py index f106c587146..5e58b7f591b 100644 --- a/tests/test_litellm/proxy/test_model_list_advertised_models.py +++ b/tests/test_litellm/proxy/test_model_list_advertised_models.py @@ -174,10 +174,10 @@ class _HidingGate(CustomLogger): def __init__(self, hidden: frozenset[str]) -> None: super().__init__() self.hidden = hidden - self.seen: list[tuple[str, ...]] = [] + self.seen: tuple[tuple[str, ...], ...] = () async def async_filter_listed_models(self, user_api_key_dict, model_names): - self.seen.append(tuple(model_names)) + self.seen = (*self.seen, tuple(model_names)) return [name for name in model_names if name not in self.hidden]