mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
* feat(bedrock): drop lookaround regex patterns from tool schemas for Converse models that reject them * fix(bedrock): rename the lookaround flag to supports_regex_lookaround and keep dropped patternProperties names allowed The cost-map flag becomes a generic supports_regex_lookaround capability, which the cost-map schema admits as a supports_* boolean, and the Converse transform now owns the drop decision instead of the shared tools factory. A patternProperties key dropped from an object closed by additionalProperties: false leaves its value schema as that object's additionalProperties, so the names it allowed stay allowed, on the OpenAI non-Python-regex drop too. tool_with_sanitized_parameters also cleans Anthropic-shape tools (input_schema). * fix(router): keep a deployment's supports_regex_lookaround off the shared cost-map entry A deployment's model_info.supports_regex_lookaround was written to the shared bedrock/<model> cost-map key, so every sibling deployment of that model id inherited one deployment's choice. The flag now stays under the deployment's own id, which is what the Converse lookaround check reads first, and the shared entry keeps the cost map's value * test(bedrock): audit the Converse lookaround drop on the wire across endpoints, SDKs, flags and chaos --------- Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
3229 lines
130 KiB
Python
3229 lines
130 KiB
Python
"""
|
|
Test that per-deployment custom pricing does not pollute the shared backend
|
|
model key in litellm.model_cost.
|
|
|
|
When two deployments share the same backend model (e.g. vertex_ai/gemini-2.5-flash)
|
|
and one has explicit zero-cost pricing in model_info, the other deployment
|
|
should still use the built-in pricing.
|
|
"""
|
|
|
|
import asyncio
|
|
import copy
|
|
import logging
|
|
import os
|
|
import re
|
|
from typing import Final
|
|
from unittest.mock import Mock, patch
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
import litellm
|
|
from litellm import Router
|
|
from litellm.caching.in_memory_cache import InMemoryCache
|
|
from litellm.constants import DEFAULT_MAX_LRU_CACHE_SIZE
|
|
from litellm.litellm_core_utils.ptu_pricing import ptu_config_error
|
|
from litellm.litellm_core_utils.llm_cost_calc.utils import SERVICE_TIER_COST_KEY_SUFFIXES
|
|
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
|
from litellm.llms.openai_like.model_info import MODEL_INFO_REFRESH_SECONDS
|
|
from litellm.types.router import Deployment, LiteLLM_Params, ModelInfo
|
|
from litellm.utils import (
|
|
_invalidate_model_cost_lowercase_map,
|
|
reapply_runtime_model_cost_registrations,
|
|
)
|
|
|
|
|
|
def _simulate_price_data_reload(fetched_catalog):
|
|
"""Drive what a price data reload does to this process's litellm state.
|
|
|
|
Mirrors `litellm.proxy.proxy_server._swap_in_model_cost_map`, which is the
|
|
one place both reload paths adopt a freshly fetched catalog; that wiring is
|
|
covered in the proxy's own tests, so these exercise the replay itself
|
|
without dragging the proxy in. The provider model sets that helper also
|
|
repopulates are left alone, since nothing here reads them and rebuilding
|
|
them from a two-entry catalog would outlive the test.
|
|
"""
|
|
litellm.model_cost = fetched_catalog
|
|
_invalidate_model_cost_lowercase_map()
|
|
reapply_runtime_model_cost_registrations()
|
|
|
|
|
|
def _nested_container_ids(value: object) -> frozenset[int]:
|
|
"""Identities of every dict/list reachable from `value`, so two structures can be
|
|
checked for shared mutable state without writing into either one."""
|
|
if isinstance(value, dict):
|
|
return frozenset({id(value)} | {i for v in value.values() for i in _nested_container_ids(v)})
|
|
if isinstance(value, list):
|
|
return frozenset({id(value)} | {i for v in value for i in _nested_container_ids(v)})
|
|
return frozenset()
|
|
|
|
|
|
def _restore_model_cost_entries(original_entries):
|
|
for key, value in original_entries.items():
|
|
if value is None:
|
|
litellm.model_cost.pop(key, None)
|
|
else:
|
|
litellm.model_cost[key] = value
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
@pytest.mark.parametrize("initial_count", (1, DEFAULT_MAX_LRU_CACHE_SIZE + 1))
|
|
async def test_discovered_limits_survive_deployment_growth_and_removal(
|
|
initial_count: int, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
|
deployments: Final = tuple(
|
|
Deployment(
|
|
model_name=f"local-{index}",
|
|
litellm_params=LiteLLM_Params(
|
|
model="hosted_vllm/local-model", api_base="https://capacity.test/v1", api_key="local-key"
|
|
),
|
|
model_info=ModelInfo(id=f"capacity-{index}"),
|
|
)
|
|
for index in range(DEFAULT_MAX_LRU_CACHE_SIZE + 2)
|
|
)
|
|
router: Final = Router(model_list=[deployment.to_json() for deployment in deployments[:initial_count]])
|
|
handler: Final = AsyncHTTPHandler()
|
|
await handler.client.aclose()
|
|
async with httpx.AsyncClient(
|
|
transport=httpx.MockTransport(
|
|
lambda request: httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": 4096}]})
|
|
)
|
|
) as client:
|
|
handler.client = client
|
|
await router.arefresh_model_info(client=handler)
|
|
assert all(
|
|
router.get_configured_token_limits(deployment.model_name) == (4096, 4096)
|
|
for deployment in deployments[:initial_count]
|
|
)
|
|
for deployment in deployments[initial_count:]:
|
|
router.add_deployment(deployment)
|
|
await router._arefresh_deployment_model_info(router.model_list[-1], client=handler)
|
|
assert all(
|
|
router.get_configured_token_limits(deployment.model_name) == (4096, 4096) for deployment in deployments
|
|
)
|
|
for deployment in deployments[-2:]:
|
|
router.delete_deployment(deployment.model_info.id or "")
|
|
await router._arefresh_deployment_model_info(router.model_list[0], client=handler)
|
|
assert all(
|
|
router.get_configured_token_limits(deployment.model_name) == (4096, 4096) for deployment in deployments[:-2]
|
|
)
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
async def test_discovery_discards_metadata_for_a_replaced_deployment(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
|
router: Final = Router(model_list=[{
|
|
"model_name": "local",
|
|
"litellm_params": {
|
|
"model": "hosted_vllm/local-model",
|
|
"api_base": "https://original.test/v1",
|
|
"api_key": "local-key",
|
|
},
|
|
"model_info": {"id": "replaced-deployment"},
|
|
}])
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
if request.url.host == "original.test":
|
|
router.upsert_deployment(Deployment(
|
|
model_name="local",
|
|
litellm_params=LiteLLM_Params(
|
|
model="hosted_vllm/local-model",
|
|
api_base="https://replacement.test/v1",
|
|
api_key="local-key",
|
|
),
|
|
model_info=ModelInfo(id="replaced-deployment"),
|
|
))
|
|
return httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": 8192}]})
|
|
assert request.url.host == "replacement.test"
|
|
return httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": 2048}]})
|
|
|
|
handler: Final = AsyncHTTPHandler()
|
|
await handler.client.aclose()
|
|
async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as client:
|
|
handler.client = client
|
|
await router._arefresh_deployment_model_info(router.model_list[0], client=handler)
|
|
assert router.get_configured_token_limits("local") == (None, None)
|
|
await router.arefresh_model_info(client=handler)
|
|
assert router.get_configured_token_limits("local") == (2048, 2048)
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
async def test_discovery_is_isolated_across_routers_and_reused_ids(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
|
first, second = tuple(
|
|
Router(model_list=[{
|
|
"model_name": "local",
|
|
"litellm_params": {
|
|
"model": "hosted_vllm/local-model",
|
|
"api_base": f"https://{host}.test/v1",
|
|
"api_key": "local-key",
|
|
},
|
|
"model_info": {"id": "shared-discovery-id"},
|
|
}])
|
|
for host in ("first", "second")
|
|
)
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
if request.url.host == "unavailable.test":
|
|
return httpx.Response(503)
|
|
limit: Final = 8192 if request.url.host == "first.test" else 2048
|
|
return httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": limit}]})
|
|
|
|
handler: Final = AsyncHTTPHandler()
|
|
await handler.client.aclose()
|
|
async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as client:
|
|
handler.client = client
|
|
await first.arefresh_model_info(client=handler)
|
|
assert second.get_configured_token_limits("local") == (None, None)
|
|
await second.arefresh_model_info(client=handler)
|
|
assert first.get_discovered_model_info("shared-discovery-id")["max_input_tokens"] == 8192
|
|
assert first.get_configured_token_limits("local") == (8192, 8192)
|
|
assert second.get_configured_token_limits("local") == (2048, 2048)
|
|
assert litellm.model_cost["shared-discovery-id"].get("max_input_tokens") is None
|
|
first.upsert_deployment(Deployment(
|
|
model_name="local",
|
|
litellm_params=LiteLLM_Params(
|
|
model="hosted_vllm/local-model",
|
|
api_base="https://unavailable.test/v1",
|
|
api_key="local-key",
|
|
),
|
|
model_info=ModelInfo(id="shared-discovery-id"),
|
|
))
|
|
assert first.get_configured_token_limits("local") == (None, None)
|
|
await first.arefresh_model_info(client=handler)
|
|
assert first.get_configured_token_limits("local") == (None, None)
|
|
assert second.get_configured_token_limits("local") == (2048, 2048)
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
async def test_discovery_refreshes_other_endpoints_while_one_is_pending(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
|
second_started: Final = asyncio.Event()
|
|
router: Final = Router(model_list=[
|
|
{
|
|
"model_name": host,
|
|
"litellm_params": {
|
|
"model": "hosted_vllm/local-model",
|
|
"api_base": f"https://{host}.test/v1",
|
|
"api_key": "local-key",
|
|
},
|
|
}
|
|
for host in ("first", "second", "third")
|
|
])
|
|
|
|
async def respond(request: httpx.Request) -> httpx.Response:
|
|
if request.url.host == "first.test":
|
|
await second_started.wait()
|
|
if request.url.host == "second.test":
|
|
second_started.set()
|
|
return httpx.Response(503)
|
|
return httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": 2048}]})
|
|
|
|
handler: Final = AsyncHTTPHandler()
|
|
await handler.client.aclose()
|
|
async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as client:
|
|
handler.client = client
|
|
await asyncio.wait_for(router.arefresh_model_info(client=handler), timeout=2)
|
|
assert router.get_configured_token_limits("first") == (2048, 2048)
|
|
assert router.get_configured_token_limits("second") == (None, None)
|
|
assert router.get_configured_token_limits("third") == (2048, 2048)
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
async def test_discovered_limits_expire_after_the_last_successful_refresh(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
|
clock: Final = Mock(return_value=0.0)
|
|
router: Final = Router(model_list=[{
|
|
"model_name": "local",
|
|
"litellm_params": {
|
|
"model": "hosted_vllm/local-model",
|
|
"api_base": "https://expiry.test/v1",
|
|
"api_key": "local-key",
|
|
},
|
|
"model_info": {"id": "expiring-discovery"},
|
|
}])
|
|
router._discovered_model_info_cache = InMemoryCache(clock=clock, default_ttl=2 * MODEL_INFO_REFRESH_SECONDS)
|
|
responses: Final = iter((
|
|
httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": 4096}]}),
|
|
httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": 8192}]}),
|
|
httpx.Response(503),
|
|
))
|
|
handler: Final = AsyncHTTPHandler()
|
|
await handler.client.aclose()
|
|
async with httpx.AsyncClient(transport=httpx.MockTransport(lambda request: next(responses))) as client:
|
|
handler.client = client
|
|
await router.arefresh_model_info(client=handler)
|
|
clock.return_value = MODEL_INFO_REFRESH_SECONDS
|
|
router.cache.in_memory_cache.flush_cache()
|
|
await router.arefresh_model_info(client=handler)
|
|
clock.return_value = 2 * MODEL_INFO_REFRESH_SECONDS + 1
|
|
router.cache.in_memory_cache.flush_cache()
|
|
await router.arefresh_model_info(client=handler)
|
|
assert router.get_configured_token_limits("local") == (8192, 8192)
|
|
group: Final = router.get_model_group_info("local")
|
|
assert group is not None
|
|
assert group.max_input_tokens == 8192
|
|
clock.return_value = 3 * MODEL_INFO_REFRESH_SECONDS + 1
|
|
await router.arefresh_model_info(client=handler)
|
|
assert router.get_configured_token_limits("local") == (None, None)
|
|
expired_group: Final = router.get_model_group_info("local")
|
|
assert expired_group is not None
|
|
assert expired_group.max_input_tokens is None
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
@pytest.mark.parametrize("provider", ("hosted_vllm", "openai", "openai_like", "text-completion-openai"))
|
|
async def test_discovered_limits_are_isolated_overridable_and_refreshable(
|
|
provider: str, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
|
upstream_limit: Final = iter((8192, 4096, 16384, 2048))
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
assert request.url.path == "/v1/models"
|
|
assert request.headers["authorization"] == "Bearer local-key"
|
|
return httpx.Response(200, json={"data": [{"id": "org/local-model", "max_model_len": next(upstream_limit)}]})
|
|
|
|
router: Final = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "local",
|
|
"litellm_params": {
|
|
"model": f"{provider}/org/local-model",
|
|
"api_base": f"https://{host}.test/v1",
|
|
"api_key": "local-key",
|
|
},
|
|
"model_info": {"id": host, **overrides},
|
|
}
|
|
for host, overrides in (("one", {}), ("two", {"max_output_tokens": 512}))
|
|
],
|
|
enable_pre_call_checks=True,
|
|
)
|
|
handler: Final = AsyncHTTPHandler()
|
|
await handler.client.aclose()
|
|
async with httpx.AsyncClient(transport=httpx.MockTransport(respond)) as client:
|
|
handler.client = client
|
|
await router.arefresh_model_info(client=handler)
|
|
first: Final = router.get_router_model_info(id="one", deployment=None, received_model_name="local")
|
|
second: Final = router.get_router_model_info(id="two", deployment=None, received_model_name="local")
|
|
assert (first["max_input_tokens"], first["max_output_tokens"]) == (8192, 8192)
|
|
assert (second["max_input_tokens"], second["max_output_tokens"]) == (4096, 512)
|
|
group: Final = router.get_model_group_info("local")
|
|
assert group is not None
|
|
assert group.max_input_tokens == 8192
|
|
listing: Final = router.get_model_listing_info("local")
|
|
assert listing is not None
|
|
assert listing.max_input_tokens == 8192
|
|
assert router.get_configured_token_limits("local") == (8192, 8192)
|
|
assert router._deployment_max_input_tokens("local", router.model_list[1]) == 4096
|
|
allowed: Final = router._pre_call_checks(
|
|
model="local", healthy_deployments=router.model_list, input="prompt", input_token_count=5000
|
|
)
|
|
assert [deployment["model_info"]["id"] for deployment in allowed] == ["one"]
|
|
assert router.model_list[0]["model_info"].get("max_input_tokens") is None
|
|
assert litellm.model_cost[f"{provider}/org/local-model"].get("max_input_tokens") is None
|
|
router.cache.in_memory_cache.flush_cache()
|
|
await router.arefresh_model_info(client=handler)
|
|
refreshed: Final = router.get_model_group_info("local")
|
|
assert refreshed is not None
|
|
assert refreshed.max_input_tokens == 16384
|
|
assert (
|
|
router.get_router_model_info(id="two", deployment=None, received_model_name="local")["max_output_tokens"]
|
|
== 512
|
|
)
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
async def test_discovery_preserves_input_overrides_and_survives_outages(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
|
responses: Final = iter((
|
|
httpx.Response(200, json={"data": [{"id": "local-model", "max_model_len": 4096}]}),
|
|
httpx.Response(503),
|
|
))
|
|
|
|
def respond(request: httpx.Request) -> httpx.Response:
|
|
assert request.url.host == "backend.test"
|
|
assert request.headers["authorization"] == "Bearer local-key"
|
|
assert request.headers["x-tenant"] == "tenant"
|
|
return next(responses)
|
|
|
|
router: Final = Router(model_list=[
|
|
{
|
|
"model_name": "configured",
|
|
"litellm_params": {
|
|
"model": "hosted_vllm/local-model",
|
|
"api_base": "https://backend.test/v1",
|
|
"api_key": "unused-key",
|
|
"extra_headers": {"authorization": "Bearer local-key", "X-Tenant": "tenant"},
|
|
},
|
|
"model_info": {"id": "configured", "max_input_tokens": 1024},
|
|
},
|
|
{
|
|
"model_name": "byok",
|
|
"litellm_params": {
|
|
"model": "openai/local-model",
|
|
"api_base": "https://caller.test/v1",
|
|
"use_clientside_credentials": True,
|
|
},
|
|
},
|
|
{"model_name": "default-openai", "litellm_params": {"model": "openai/local-model", "api_key": "unused"}},
|
|
])
|
|
handler: Final = AsyncHTTPHandler()
|
|
await handler.client.aclose()
|
|
responder: Final = Mock(side_effect=respond)
|
|
async with httpx.AsyncClient(transport=httpx.MockTransport(responder)) as client:
|
|
handler.client = client
|
|
await router.arefresh_model_info(client=handler)
|
|
assert router.get_configured_token_limits("configured") == (1024, 4096)
|
|
router.cache.in_memory_cache.flush_cache()
|
|
await router.arefresh_model_info(client=handler)
|
|
assert router.get_configured_token_limits("configured") == (1024, 4096)
|
|
assert router.get_configured_token_limits("byok") == (None, None)
|
|
assert next(responses, None) is None
|
|
assert responder.call_count == 2
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_should_not_pollute_shared_key_with_zero_cost_pricing():
|
|
"""
|
|
When deployment A has input_cost_per_token=0 and deployment B has no
|
|
custom pricing, deployment B should still report the built-in pricing
|
|
(not zero).
|
|
"""
|
|
backend_model = "vertex_ai/gemini-2.5-flash"
|
|
|
|
# Grab built-in pricing before creating any router
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
builtin_input_cost = builtin_info["input_cost_per_token"]
|
|
builtin_output_cost = builtin_info["output_cost_per_token"]
|
|
|
|
# Sanity: built-in pricing should be non-zero for this model
|
|
assert builtin_input_cost > 0, "Test requires a model with non-zero built-in pricing"
|
|
assert builtin_output_cost > 0, "Test requires a model with non-zero built-in pricing"
|
|
|
|
router = Router(
|
|
model_list=[
|
|
# Deployment A: explicit zero-cost pricing
|
|
{
|
|
"model_name": "custom-zero-cost-model",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-1",
|
|
},
|
|
"model_info": {
|
|
"id": "deployment-a-zero-cost",
|
|
"input_cost_per_token": 0.0,
|
|
"output_cost_per_token": 0.0,
|
|
},
|
|
},
|
|
# Deployment B: no custom pricing, relies on built-in
|
|
{
|
|
"model_name": "standard-cost-model",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-2",
|
|
},
|
|
"model_info": {
|
|
"id": "deployment-b-builtin-cost",
|
|
},
|
|
},
|
|
],
|
|
)
|
|
|
|
# Deployment A: should report zero pricing via its unique model_id
|
|
info_a = router.get_deployment_model_info(
|
|
model_id="deployment-a-zero-cost",
|
|
model_name=backend_model,
|
|
)
|
|
assert info_a is not None
|
|
assert info_a["input_cost_per_token"] == 0.0
|
|
assert info_a["output_cost_per_token"] == 0.0
|
|
|
|
# Deployment B: should report built-in pricing, NOT zero
|
|
info_b = router.get_deployment_model_info(
|
|
model_id="deployment-b-builtin-cost",
|
|
model_name=backend_model,
|
|
)
|
|
assert info_b is not None
|
|
assert info_b["input_cost_per_token"] == builtin_input_cost, (
|
|
f"Deployment B should use built-in input cost {builtin_input_cost}, got {info_b['input_cost_per_token']}"
|
|
)
|
|
assert info_b["output_cost_per_token"] == builtin_output_cost, (
|
|
f"Deployment B should use built-in output cost {builtin_output_cost}, got {info_b['output_cost_per_token']}"
|
|
)
|
|
|
|
|
|
def test_should_not_pollute_shared_key_with_custom_nonzero_pricing():
|
|
"""
|
|
A deployment with custom (non-zero) pricing should not overwrite
|
|
the shared backend key's built-in pricing.
|
|
"""
|
|
backend_model = "vertex_ai/gemini-2.5-flash"
|
|
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
builtin_input_cost = builtin_info["input_cost_per_token"]
|
|
|
|
router = Router(
|
|
model_list=[
|
|
# Deployment with custom high pricing
|
|
{
|
|
"model_name": "expensive-model",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-3",
|
|
},
|
|
"model_info": {
|
|
"id": "deployment-expensive",
|
|
"input_cost_per_token": 0.99,
|
|
"output_cost_per_token": 0.99,
|
|
},
|
|
},
|
|
# Deployment relying on built-in pricing
|
|
{
|
|
"model_name": "standard-model",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-4",
|
|
},
|
|
"model_info": {
|
|
"id": "deployment-standard",
|
|
},
|
|
},
|
|
],
|
|
)
|
|
|
|
# Custom pricing deployment should see its custom values
|
|
info_expensive = router.get_deployment_model_info(
|
|
model_id="deployment-expensive",
|
|
model_name=backend_model,
|
|
)
|
|
assert info_expensive is not None
|
|
assert info_expensive["input_cost_per_token"] == 0.99
|
|
assert info_expensive["output_cost_per_token"] == 0.99
|
|
|
|
# Standard deployment should still see built-in pricing
|
|
info_standard = router.get_deployment_model_info(
|
|
model_id="deployment-standard",
|
|
model_name=backend_model,
|
|
)
|
|
assert info_standard is not None
|
|
assert info_standard["input_cost_per_token"] == builtin_input_cost, (
|
|
f"Standard deployment should use built-in pricing {builtin_input_cost}, "
|
|
f"got {info_standard['input_cost_per_token']}"
|
|
)
|
|
|
|
|
|
def test_regex_lookaround_flag_stays_on_the_deployment_that_set_it() -> None:
|
|
"""A deployment's ``supports_regex_lookaround`` override must not land on the shared
|
|
``{provider}/{model}`` key, or every sibling deployment of that model would inherit it."""
|
|
backend_model = "bedrock/us.xai.grok-4.6"
|
|
deploy_id = "grok-deploy-keep-regex"
|
|
|
|
builtin_flag = litellm.get_model_info(model=backend_model).get("supports_regex_lookaround")
|
|
model_keys = {
|
|
deploy_id: litellm.model_cost.get(deploy_id),
|
|
backend_model: copy.deepcopy(litellm.model_cost.get(backend_model)),
|
|
}
|
|
try:
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "grok-keep-regex",
|
|
"litellm_params": {"model": backend_model},
|
|
"model_info": {"id": deploy_id, "supports_regex_lookaround": not builtin_flag},
|
|
}
|
|
],
|
|
)
|
|
|
|
assert litellm.model_cost[deploy_id]["supports_regex_lookaround"] is (not builtin_flag)
|
|
assert litellm.get_model_info(model=backend_model).get("supports_regex_lookaround") is builtin_flag
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_should_store_full_pricing_under_deployment_model_id():
|
|
"""
|
|
Per-deployment pricing (including zero) should be stored and
|
|
retrievable via the unique model_id key in litellm.model_cost.
|
|
"""
|
|
backend_model = "vertex_ai/gemini-2.5-flash"
|
|
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "zero-cost-model",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-5",
|
|
},
|
|
"model_info": {
|
|
"id": "deployment-zero-check",
|
|
"input_cost_per_token": 0.0,
|
|
"output_cost_per_token": 0.0,
|
|
},
|
|
},
|
|
],
|
|
)
|
|
|
|
# The model_id entry should exist and have the zero pricing
|
|
entry = litellm.model_cost.get("deployment-zero-check")
|
|
assert entry is not None, "Deployment should be registered by model_id"
|
|
assert entry["input_cost_per_token"] == 0.0
|
|
assert entry["output_cost_per_token"] == 0.0
|
|
|
|
|
|
def test_should_preserve_builtin_pricing_regardless_of_deployment_order():
|
|
"""
|
|
The built-in pricing should be preserved no matter which deployment
|
|
is processed first (zero-cost first, or standard first).
|
|
"""
|
|
backend_model = "vertex_ai/gemini-2.5-flash"
|
|
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
builtin_input_cost = builtin_info["input_cost_per_token"]
|
|
builtin_output_cost = builtin_info["output_cost_per_token"]
|
|
|
|
# Order 1: standard first, then zero-cost
|
|
router1 = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "standard-first",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-6",
|
|
},
|
|
"model_info": {"id": "order1-standard"},
|
|
},
|
|
{
|
|
"model_name": "zero-cost-second",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-7",
|
|
},
|
|
"model_info": {
|
|
"id": "order1-zero",
|
|
"input_cost_per_token": 0.0,
|
|
"output_cost_per_token": 0.0,
|
|
},
|
|
},
|
|
],
|
|
)
|
|
|
|
info_std_1 = router1.get_deployment_model_info(model_id="order1-standard", model_name=backend_model)
|
|
assert info_std_1["input_cost_per_token"] == builtin_input_cost
|
|
assert info_std_1["output_cost_per_token"] == builtin_output_cost
|
|
|
|
# Order 2: zero-cost first, then standard
|
|
router2 = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "zero-cost-first",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-8",
|
|
},
|
|
"model_info": {
|
|
"id": "order2-zero",
|
|
"input_cost_per_token": 0.0,
|
|
"output_cost_per_token": 0.0,
|
|
},
|
|
},
|
|
{
|
|
"model_name": "standard-second",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-9",
|
|
},
|
|
"model_info": {"id": "order2-standard"},
|
|
},
|
|
],
|
|
)
|
|
|
|
info_std_2 = router2.get_deployment_model_info(model_id="order2-standard", model_name=backend_model)
|
|
assert info_std_2["input_cost_per_token"] == builtin_input_cost, (
|
|
f"Order should not matter. Expected {builtin_input_cost}, got {info_std_2['input_cost_per_token']}"
|
|
)
|
|
assert info_std_2["output_cost_per_token"] == builtin_output_cost, (
|
|
f"Order should not matter. Expected {builtin_output_cost}, got {info_std_2['output_cost_per_token']}"
|
|
)
|
|
|
|
|
|
def test_responses_prefix_stripped_alias_registered_for_model_list():
|
|
"""
|
|
Register ``litellm.model_cost`` under the backend key with ``responses/`` and
|
|
under the stripped key (``responses_api_bridge_check`` removes that segment).
|
|
"""
|
|
uid = "responses-strip-alias-test-a1b2c3d4"
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "azure-responses-strip-test",
|
|
"litellm_params": {
|
|
"model": "responses/gpt-strip-test-a1b2c3d4",
|
|
"custom_llm_provider": "azure",
|
|
"api_key": "fake-key-strip",
|
|
},
|
|
"model_info": {
|
|
"id": uid,
|
|
"supports_native_streaming": True,
|
|
},
|
|
}
|
|
],
|
|
)
|
|
assert "azure/responses/gpt-strip-test-a1b2c3d4" in litellm.model_cost
|
|
assert "azure/gpt-strip-test-a1b2c3d4" in litellm.model_cost
|
|
assert litellm.model_cost["azure/gpt-strip-test-a1b2c3d4"].get("supports_native_streaming") is True
|
|
|
|
|
|
def test_responses_prefix_stripped_alias_registered_for_add_deployment():
|
|
"""Dynamic ``add_deployment`` must mirror ``_create_deployment`` registration."""
|
|
uid = "add-dep-responses-strip-e5f6a7b8"
|
|
router = Router(model_list=[])
|
|
deployment = Deployment(
|
|
model_name="dyn-responses-strip",
|
|
litellm_params=LiteLLM_Params(
|
|
model="responses/gpt-add-strip-e5f6a7b8",
|
|
custom_llm_provider="azure",
|
|
api_key="fake-key-add",
|
|
),
|
|
model_info=ModelInfo(id=uid, supports_native_streaming=True),
|
|
)
|
|
router.add_deployment(deployment=deployment)
|
|
assert "azure/responses/gpt-add-strip-e5f6a7b8" in litellm.model_cost
|
|
assert "azure/gpt-add-strip-e5f6a7b8" in litellm.model_cost
|
|
assert litellm.model_cost["azure/gpt-add-strip-e5f6a7b8"].get("supports_native_streaming") is True
|
|
|
|
|
|
def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override():
|
|
"""
|
|
ChatGPT aliases that share the same backend model should not be able to
|
|
downgrade the shared backend key from responses -> chat during router setup.
|
|
"""
|
|
from litellm.main import responses_api_bridge_check
|
|
|
|
backend_model = "chatgpt/gpt-5.4"
|
|
model_keys = {
|
|
backend_model: copy.deepcopy(litellm.model_cost.get(backend_model)),
|
|
"chatgpt-shared-mode-base": copy.deepcopy(litellm.model_cost.get("chatgpt-shared-mode-base")),
|
|
"chatgpt-shared-mode-alias": copy.deepcopy(litellm.model_cost.get("chatgpt-shared-mode-alias")),
|
|
}
|
|
|
|
try:
|
|
backend_entry = copy.deepcopy(model_keys[backend_model]) or {}
|
|
backend_entry["litellm_provider"] = "chatgpt"
|
|
backend_entry["mode"] = "responses"
|
|
litellm.model_cost[backend_model] = backend_entry
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
router = Router(model_list=[])
|
|
with patch.object(Router, "_add_deployment", lambda self, deployment: deployment):
|
|
router._create_deployment(
|
|
deployment_info={},
|
|
_model_name="chatgpt/gpt-5.4",
|
|
_litellm_params={
|
|
"model": "gpt-5.4",
|
|
"custom_llm_provider": "chatgpt",
|
|
},
|
|
_model_info={
|
|
"id": "chatgpt-shared-mode-base",
|
|
"mode": "responses",
|
|
},
|
|
)
|
|
router._create_deployment(
|
|
deployment_info={},
|
|
_model_name="chatgpt/gpt-5.4-medium",
|
|
_litellm_params={
|
|
"model": "gpt-5.4",
|
|
"custom_llm_provider": "chatgpt",
|
|
},
|
|
_model_info={
|
|
"id": "chatgpt-shared-mode-alias",
|
|
"mode": "chat",
|
|
},
|
|
)
|
|
|
|
assert litellm.model_cost[backend_model]["mode"] == "responses"
|
|
assert "mode" in litellm.model_cost[backend_model]
|
|
|
|
bridge_model_info, bridge_model = responses_api_bridge_check(
|
|
model="gpt-5.4",
|
|
custom_llm_provider="chatgpt",
|
|
)
|
|
assert bridge_model == "gpt-5.4"
|
|
assert bridge_model_info["mode"] == "responses"
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_partial_custom_pricing_inherits_builtin_cache_pricing():
|
|
"""A deployment that overrides only input/output cost on a cache-supporting
|
|
model must still bill cache_read and cache_creation tokens. Before the
|
|
fix the deploy-id entry was registered with the user's two fields and
|
|
nothing else, so the cost calculator silently billed cache tokens at 0.
|
|
Regression for the prompt-caching cost dropout reported by the customer.
|
|
"""
|
|
backend_model = "anthropic/claude-sonnet-4-5-20250929"
|
|
deploy_id = "claude-deploy-partial-pricing"
|
|
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
builtin_cache_create = builtin_info["cache_creation_input_token_cost"]
|
|
builtin_cache_read = builtin_info["cache_read_input_token_cost"]
|
|
assert builtin_cache_create is not None and builtin_cache_create > 0
|
|
assert builtin_cache_read is not None and builtin_cache_read > 0
|
|
|
|
model_keys = {
|
|
deploy_id: litellm.model_cost.get(deploy_id),
|
|
backend_model: copy.deepcopy(litellm.model_cost.get(backend_model)),
|
|
}
|
|
try:
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "claude-custom",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key",
|
|
},
|
|
"model_info": {
|
|
"id": deploy_id,
|
|
"input_cost_per_token": 0.000003,
|
|
"output_cost_per_token": 0.000015,
|
|
},
|
|
}
|
|
],
|
|
)
|
|
|
|
entry = litellm.model_cost[deploy_id]
|
|
assert entry["input_cost_per_token"] == 0.000003
|
|
assert entry["output_cost_per_token"] == 0.000015
|
|
assert entry.get("cache_creation_input_token_cost") == builtin_cache_create
|
|
assert entry.get("cache_read_input_token_cost") == builtin_cache_read
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_partial_pricing_does_not_overwrite_explicit_cache_fields():
|
|
"""When the user explicitly sets cache_*_input_token_cost on a deployment,
|
|
those values must not be replaced by the built-in fallback.
|
|
"""
|
|
backend_model = "anthropic/claude-sonnet-4-5-20250929"
|
|
deploy_id = "claude-deploy-explicit-cache"
|
|
|
|
explicit_cache_create = 0.00001
|
|
explicit_cache_read = 0.0000005
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
assert builtin_info["cache_creation_input_token_cost"] != explicit_cache_create
|
|
assert builtin_info["cache_read_input_token_cost"] != explicit_cache_read
|
|
|
|
model_keys = {
|
|
deploy_id: litellm.model_cost.get(deploy_id),
|
|
backend_model: copy.deepcopy(litellm.model_cost.get(backend_model)),
|
|
}
|
|
try:
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "claude-custom-explicit",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key",
|
|
},
|
|
"model_info": {
|
|
"id": deploy_id,
|
|
"input_cost_per_token": 0.000003,
|
|
"output_cost_per_token": 0.000015,
|
|
"cache_creation_input_token_cost": explicit_cache_create,
|
|
"cache_read_input_token_cost": explicit_cache_read,
|
|
},
|
|
}
|
|
],
|
|
)
|
|
|
|
entry = litellm.model_cost[deploy_id]
|
|
assert entry.get("cache_creation_input_token_cost") == explicit_cache_create
|
|
assert entry.get("cache_read_input_token_cost") == explicit_cache_read
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_inherit_builtin_cache_pricing_fills_only_missing_fields():
|
|
"""Direct unit test of the helper: missing cache fields are filled from the
|
|
backend model's built-in entry, while an explicitly set cache field and the
|
|
user's input/output pricing are left untouched.
|
|
"""
|
|
backend_model = "anthropic/claude-sonnet-4-5-20250929"
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
builtin_cache_create = builtin_info["cache_creation_input_token_cost"]
|
|
builtin_cache_read = builtin_info["cache_read_input_token_cost"]
|
|
assert builtin_cache_create is not None and builtin_cache_create > 0
|
|
assert builtin_cache_read is not None and builtin_cache_read > 0
|
|
|
|
explicit_cache_read = builtin_cache_read + 1
|
|
model_info = {
|
|
"input_cost_per_token": 0.000003,
|
|
"cache_read_input_token_cost": explicit_cache_read,
|
|
}
|
|
|
|
Router._inherit_builtin_cache_pricing(
|
|
model_info=model_info,
|
|
backend_model=backend_model,
|
|
custom_llm_provider="anthropic",
|
|
)
|
|
|
|
assert model_info["input_cost_per_token"] == 0.000003
|
|
assert model_info["cache_read_input_token_cost"] == explicit_cache_read
|
|
assert model_info["cache_creation_input_token_cost"] == builtin_cache_create
|
|
|
|
|
|
def test_inherit_builtin_cache_pricing_noop_for_unknown_backend():
|
|
"""No canonical entry for the backend model means the helper leaves the
|
|
passed-in dict unchanged rather than raising.
|
|
"""
|
|
model_info = {"input_cost_per_token": 0.000003}
|
|
|
|
Router._inherit_builtin_cache_pricing(
|
|
model_info=model_info,
|
|
backend_model="this-backend-model-does-not-exist-x9y8z7",
|
|
custom_llm_provider=None,
|
|
)
|
|
|
|
assert model_info == {"input_cost_per_token": 0.000003}
|
|
|
|
|
|
_TIER_BACKEND_MODEL: Final = "tier-priced-backend"
|
|
_TIER_BACKEND_KEY: Final = f"openai/{_TIER_BACKEND_MODEL}"
|
|
_CUSTOM_STANDARD_INPUT_RATE: Final = 0.00011
|
|
_CUSTOM_STANDARD_OUTPUT_RATE: Final = 0.00022
|
|
_TIER_BACKEND_ENTRY: Final = {
|
|
"key": _TIER_BACKEND_KEY,
|
|
"litellm_provider": "openai",
|
|
"mode": "chat",
|
|
"max_tokens": 123456,
|
|
"input_cost_per_token": 0.00021,
|
|
"output_cost_per_token": 0.00032,
|
|
"input_cost_per_token_ultrafast": 0.00031,
|
|
"output_cost_per_token_ultrafast": 0.00042,
|
|
"input_cost_per_token_priority": 0.00051,
|
|
"output_cost_per_token_priority": 0.00062,
|
|
"input_cost_per_token_flex": 0.00071,
|
|
"output_cost_per_token_flex": 0.00082,
|
|
"input_cost_per_token_balanced": 0.00091,
|
|
"output_cost_per_token_balanced": 0.00102,
|
|
"cache_read_input_token_cost_ultrafast": 0.00013,
|
|
"input_cost_per_token_above_272k_tokens_ultrafast": 0.00014,
|
|
"output_cost_per_token_above_272k_tokens_ultrafast": 0.00015,
|
|
"input_cost_per_token_batches": 0.00016,
|
|
"input_cost_per_token_above_272k_tokens": 0.00017,
|
|
}
|
|
_AZURE_TIER_BACKEND_KEY: Final = "azure/tier-priced-backend"
|
|
_AZURE_TIER_BACKEND_ENTRY: Final = {
|
|
**_TIER_BACKEND_ENTRY,
|
|
"key": _AZURE_TIER_BACKEND_KEY,
|
|
"litellm_provider": "azure",
|
|
}
|
|
|
|
|
|
def _register_tier_backend() -> None:
|
|
litellm.model_cost[_TIER_BACKEND_KEY] = copy.deepcopy(_TIER_BACKEND_ENTRY)
|
|
litellm.get_model_info.cache_clear()
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def _register_azure_tier_backend() -> None:
|
|
litellm.model_cost[_AZURE_TIER_BACKEND_KEY] = copy.deepcopy(_AZURE_TIER_BACKEND_ENTRY)
|
|
litellm.get_model_info.cache_clear()
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_inherit_builtin_service_tier_pricing_fills_only_missing_fields() -> None:
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL)
|
|
}
|
|
try:
|
|
_register_tier_backend()
|
|
model_info: Final = {
|
|
"id": "custom-priced-tier-deployment",
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
"output_cost_per_token_ultrafast": 0.00999,
|
|
}
|
|
|
|
Router._inherit_builtin_service_tier_pricing(
|
|
model_info=model_info,
|
|
backend_model=_TIER_BACKEND_MODEL,
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model_info == {
|
|
"id": "custom-priced-tier-deployment",
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
"input_cost_per_token_ultrafast": _TIER_BACKEND_ENTRY["input_cost_per_token_ultrafast"],
|
|
"output_cost_per_token_ultrafast": 0.00999,
|
|
"input_cost_per_token_priority": _TIER_BACKEND_ENTRY["input_cost_per_token_priority"],
|
|
"output_cost_per_token_priority": _TIER_BACKEND_ENTRY["output_cost_per_token_priority"],
|
|
"input_cost_per_token_flex": _TIER_BACKEND_ENTRY["input_cost_per_token_flex"],
|
|
"output_cost_per_token_flex": _TIER_BACKEND_ENTRY["output_cost_per_token_flex"],
|
|
"input_cost_per_token_balanced": _TIER_BACKEND_ENTRY["input_cost_per_token_balanced"],
|
|
"output_cost_per_token_balanced": _TIER_BACKEND_ENTRY["output_cost_per_token_balanced"],
|
|
"cache_read_input_token_cost_ultrafast": _TIER_BACKEND_ENTRY[
|
|
"cache_read_input_token_cost_ultrafast"
|
|
],
|
|
"input_cost_per_token_above_272k_tokens_ultrafast": _TIER_BACKEND_ENTRY[
|
|
"input_cost_per_token_above_272k_tokens_ultrafast"
|
|
],
|
|
"output_cost_per_token_above_272k_tokens_ultrafast": _TIER_BACKEND_ENTRY[
|
|
"output_cost_per_token_above_272k_tokens_ultrafast"
|
|
],
|
|
}
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_inherit_builtin_service_tier_pricing_noop_without_base_rate_or_backend() -> None:
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL)
|
|
}
|
|
try:
|
|
_register_tier_backend()
|
|
model_info_without_base_rate: Final = {
|
|
"id": "custom-priced-no-base-rate",
|
|
"input_cost_per_token_ultrafast": 0.00031,
|
|
}
|
|
expected_without_base_rate: Final = copy.deepcopy(model_info_without_base_rate)
|
|
Router._inherit_builtin_service_tier_pricing(
|
|
model_info=model_info_without_base_rate,
|
|
backend_model=_TIER_BACKEND_MODEL,
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
model_info_with_unknown_backend: Final = {
|
|
"id": "custom-priced-unknown-backend",
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
}
|
|
expected_with_unknown_backend: Final = copy.deepcopy(model_info_with_unknown_backend)
|
|
Router._inherit_builtin_service_tier_pricing(
|
|
model_info=model_info_with_unknown_backend,
|
|
backend_model="tier-priced-backend-unknown",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model_info_without_base_rate == expected_without_base_rate
|
|
assert model_info_with_unknown_backend == expected_with_unknown_backend
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_router_completion_uses_custom_standard_and_backend_ultrafast_pricing() -> None:
|
|
model_id: Final = "tier-priced-deployment"
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL, model_id)
|
|
}
|
|
try:
|
|
_register_tier_backend()
|
|
router: Final = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "tier-priced-router",
|
|
"litellm_params": {
|
|
"model": _TIER_BACKEND_MODEL,
|
|
"custom_llm_provider": "openai",
|
|
"api_key": "sk-tier-pricing-not-used",
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
},
|
|
"model_info": {
|
|
"id": model_id,
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
},
|
|
}
|
|
]
|
|
)
|
|
|
|
ultrafast_response: Final = router.completion(
|
|
model="tier-priced-router",
|
|
messages=[{"role": "user", "content": "tiered pricing"}],
|
|
service_tier="ultrafast",
|
|
mock_response=litellm.ModelResponse(
|
|
model=_TIER_BACKEND_MODEL,
|
|
service_tier="ultrafast",
|
|
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
|
),
|
|
)
|
|
standard_response: Final = router.completion(
|
|
model="tier-priced-router",
|
|
messages=[{"role": "user", "content": "standard pricing"}],
|
|
mock_response=litellm.ModelResponse(
|
|
model=_TIER_BACKEND_MODEL,
|
|
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
|
),
|
|
)
|
|
|
|
assert isinstance(ultrafast_response, litellm.ModelResponse)
|
|
assert ultrafast_response._hidden_params["response_cost"] == pytest.approx(
|
|
1000 * _TIER_BACKEND_ENTRY["input_cost_per_token_ultrafast"]
|
|
+ 100 * _TIER_BACKEND_ENTRY["output_cost_per_token_ultrafast"]
|
|
)
|
|
assert isinstance(standard_response, litellm.ModelResponse)
|
|
assert standard_response._hidden_params["response_cost"] == pytest.approx(
|
|
1000 * _CUSTOM_STANDARD_INPUT_RATE + 100 * _CUSTOM_STANDARD_OUTPUT_RATE
|
|
)
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_router_completion_uses_backend_ultrafast_long_context_rates() -> None:
|
|
model_id: Final = "tier-priced-long-context-deployment"
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL, model_id)
|
|
}
|
|
try:
|
|
_register_tier_backend()
|
|
router: Final = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "tier-priced-long-context-router",
|
|
"litellm_params": {
|
|
"model": _TIER_BACKEND_MODEL,
|
|
"custom_llm_provider": "openai",
|
|
"api_key": "sk-tier-pricing-not-used",
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
},
|
|
"model_info": {
|
|
"id": model_id,
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
},
|
|
}
|
|
]
|
|
)
|
|
|
|
response: Final = router.completion(
|
|
model="tier-priced-long-context-router",
|
|
messages=[{"role": "user", "content": "long context tiered pricing"}],
|
|
service_tier="ultrafast",
|
|
mock_response=litellm.ModelResponse(
|
|
model=_TIER_BACKEND_MODEL,
|
|
service_tier="ultrafast",
|
|
usage=litellm.Usage(prompt_tokens=300_000, completion_tokens=100, total_tokens=300_100),
|
|
),
|
|
)
|
|
|
|
assert isinstance(response, litellm.ModelResponse)
|
|
assert response._hidden_params["response_cost"] == pytest.approx(
|
|
300_000 * _TIER_BACKEND_ENTRY["input_cost_per_token_above_272k_tokens_ultrafast"]
|
|
+ 100 * _TIER_BACKEND_ENTRY["output_cost_per_token_above_272k_tokens_ultrafast"]
|
|
)
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
@pytest.mark.parametrize("ptu_enabled", (True, False))
|
|
def test_ptu_service_tier_pricing_is_disabled_only_when_attribution_is_enabled(
|
|
monkeypatch: pytest.MonkeyPatch, ptu_enabled: bool
|
|
) -> None:
|
|
model_id: Final = f"ptu-tier-deployment-{ptu_enabled}"
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (_TIER_BACKEND_KEY, model_id)
|
|
}
|
|
try:
|
|
_register_tier_backend()
|
|
monkeypatch.setenv("LITELLM_ENABLE_PTU_COST_ATTRIBUTION", "True" if ptu_enabled else "")
|
|
router: Final = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": f"ptu-tier-model-{ptu_enabled}",
|
|
"litellm_params": {
|
|
"model": _TIER_BACKEND_MODEL,
|
|
"custom_llm_provider": "openai",
|
|
"api_key": "sk-tier-pricing-not-used",
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
},
|
|
"model_info": {**_PTU_MODEL_INFO, "id": model_id},
|
|
}
|
|
]
|
|
)
|
|
registered: Final = litellm.model_cost[model_id]
|
|
tier_fields: Final = tuple(
|
|
field for field in _TIER_BACKEND_ENTRY if field.endswith(SERVICE_TIER_COST_KEY_SUFFIXES)
|
|
)
|
|
if ptu_enabled:
|
|
assert all(field not in registered for field in tier_fields)
|
|
else:
|
|
assert all(field in registered for field in tier_fields)
|
|
|
|
response: Final = router.completion(
|
|
model=f"ptu-tier-model-{ptu_enabled}",
|
|
messages=[{"role": "user", "content": "ptu service tier pricing"}],
|
|
service_tier="priority",
|
|
mock_response=litellm.ModelResponse(
|
|
model=_TIER_BACKEND_MODEL,
|
|
service_tier="priority",
|
|
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
|
),
|
|
)
|
|
|
|
assert isinstance(response, litellm.ModelResponse)
|
|
expected_cost: Final = (
|
|
0.0
|
|
if ptu_enabled
|
|
else 1000 * _TIER_BACKEND_ENTRY["input_cost_per_token_priority"]
|
|
+ 100 * _TIER_BACKEND_ENTRY["output_cost_per_token_priority"]
|
|
)
|
|
assert response._hidden_params["response_cost"] == pytest.approx(expected_cost)
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_azure_base_model_inherits_service_tier_pricing_for_registration_and_payload() -> None:
|
|
model_id: Final = "azure-tier-priced-alias"
|
|
payload_id: Final = "azure-tier-priced-payload"
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (_AZURE_TIER_BACKEND_KEY, model_id, payload_id)
|
|
}
|
|
try:
|
|
_register_azure_tier_backend()
|
|
router: Final = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "azure/tier-priced-alias",
|
|
"litellm_params": {
|
|
"model": "azure/tier-priced-alias",
|
|
"custom_llm_provider": "azure",
|
|
"api_key": "sk-tier-pricing-not-used",
|
|
"api_base": "https://tier-priced.azure.invalid",
|
|
},
|
|
"model_info": {
|
|
"id": model_id,
|
|
"base_model": _AZURE_TIER_BACKEND_KEY,
|
|
"input_cost_per_token": _CUSTOM_STANDARD_INPUT_RATE,
|
|
"output_cost_per_token": _CUSTOM_STANDARD_OUTPUT_RATE,
|
|
},
|
|
}
|
|
]
|
|
)
|
|
|
|
response: Final = router.completion(
|
|
model="azure/tier-priced-alias",
|
|
messages=[{"role": "user", "content": "azure base model pricing"}],
|
|
service_tier="priority",
|
|
allowed_openai_params=["service_tier"],
|
|
mock_response=litellm.ModelResponse(
|
|
model=_AZURE_TIER_BACKEND_KEY,
|
|
service_tier="priority",
|
|
usage=litellm.Usage(prompt_tokens=1000, completion_tokens=100, total_tokens=1100),
|
|
),
|
|
)
|
|
|
|
assert isinstance(response, litellm.ModelResponse)
|
|
assert response._hidden_params["response_cost"] == pytest.approx(
|
|
1000 * _AZURE_TIER_BACKEND_ENTRY["input_cost_per_token_priority"]
|
|
+ 100 * _AZURE_TIER_BACKEND_ENTRY["output_cost_per_token_priority"]
|
|
)
|
|
|
|
payload: Final = Router._deployment_model_cost_payload(
|
|
deployment=Deployment(
|
|
model_name="azure/tier-priced-alias-from-params",
|
|
litellm_params=LiteLLM_Params(
|
|
model="azure/tier-priced-alias",
|
|
custom_llm_provider="azure",
|
|
base_model=_AZURE_TIER_BACKEND_KEY,
|
|
input_cost_per_token=_CUSTOM_STANDARD_INPUT_RATE,
|
|
output_cost_per_token=_CUSTOM_STANDARD_OUTPUT_RATE,
|
|
),
|
|
model_info=ModelInfo(id=payload_id),
|
|
)
|
|
)
|
|
|
|
assert payload["input_cost_per_token_priority"] == _AZURE_TIER_BACKEND_ENTRY[
|
|
"input_cost_per_token_priority"
|
|
]
|
|
assert payload["output_cost_per_token_priority"] == _AZURE_TIER_BACKEND_ENTRY[
|
|
"output_cost_per_token_priority"
|
|
]
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("model_info_base_model", "params_base_model", "model", "expected"),
|
|
(
|
|
pytest.param(
|
|
"azure/tier-priced-model-info-base",
|
|
"azure/tier-priced-params-base",
|
|
"azure/tier-priced-deployment-alias",
|
|
"azure/tier-priced-model-info-base",
|
|
id="model-info-base-model-wins",
|
|
),
|
|
pytest.param(
|
|
None,
|
|
"azure/tier-priced-params-base",
|
|
"azure/tier-priced-deployment-alias",
|
|
"azure/tier-priced-params-base",
|
|
id="params-base-model-fallback",
|
|
),
|
|
pytest.param(
|
|
None,
|
|
None,
|
|
"azure/tier-priced-deployment-alias",
|
|
"azure/tier-priced-deployment-alias",
|
|
id="model-fallback",
|
|
),
|
|
pytest.param(
|
|
"",
|
|
"azure/tier-priced-params-base",
|
|
"azure/tier-priced-deployment-alias",
|
|
"azure/tier-priced-params-base",
|
|
id="empty-model-info-base-model-falls-through",
|
|
),
|
|
),
|
|
)
|
|
def test_cost_map_backend_model_uses_canonical_model_precedence(
|
|
model_info_base_model: str | None,
|
|
params_base_model: str | None,
|
|
model: str,
|
|
expected: str,
|
|
) -> None:
|
|
deployment: Final = Deployment(
|
|
model_name="azure/tier-priced-cost-map-backend",
|
|
litellm_params=LiteLLM_Params(model=model, base_model=params_base_model),
|
|
model_info=ModelInfo(id="tier-priced-cost-map-backend", base_model=model_info_base_model),
|
|
)
|
|
|
|
assert Router._cost_map_backend_model(deployment) == expected
|
|
|
|
|
|
def test_inherit_builtin_base_rates_for_off_peak_fills_missing_rates():
|
|
"""Direct unit test of the helper: an entry carrying only an
|
|
off_peak_pricing block inherits the backend model's built-in base token
|
|
rates, so cost lookup via the deployment id can bill standard rates
|
|
outside the windows.
|
|
"""
|
|
backend_model = "gpt-4o-mini"
|
|
builtin_info = litellm.get_model_info(model=backend_model, custom_llm_provider="openai")
|
|
off_peak_block = {
|
|
"hours_utc": "00:00-00:00",
|
|
"input_cost_per_token": 5e-07,
|
|
"output_cost_per_token": 1e-06,
|
|
}
|
|
model_info = {"off_peak_pricing": off_peak_block}
|
|
|
|
Router._inherit_builtin_base_rates_for_off_peak(
|
|
model_info=model_info,
|
|
backend_model=backend_model,
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model_info["input_cost_per_token"] == builtin_info["input_cost_per_token"]
|
|
assert model_info["output_cost_per_token"] == builtin_info["output_cost_per_token"]
|
|
assert model_info["off_peak_pricing"] == off_peak_block
|
|
|
|
|
|
def test_inherit_builtin_base_rates_for_off_peak_carries_threshold_rates():
|
|
"""A backend with above-threshold pricing hands the whole rate structure to
|
|
the deployment entry, so peak-hour billing of large prompts through that
|
|
entry matches the shared backend entry instead of flattening to the base
|
|
rate.
|
|
"""
|
|
backend_model = "gemini/gemini-2.5-pro"
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
assert builtin_info["input_cost_per_token_above_200k_tokens"] is not None
|
|
|
|
model_info = {
|
|
"off_peak_pricing": {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-07},
|
|
}
|
|
|
|
Router._inherit_builtin_base_rates_for_off_peak(
|
|
model_info=model_info,
|
|
backend_model=backend_model,
|
|
custom_llm_provider="gemini",
|
|
)
|
|
|
|
assert model_info["input_cost_per_token"] == builtin_info["input_cost_per_token"]
|
|
assert (
|
|
model_info["input_cost_per_token_above_200k_tokens"]
|
|
== builtin_info["input_cost_per_token_above_200k_tokens"]
|
|
)
|
|
assert (
|
|
model_info["output_cost_per_token_above_200k_tokens"]
|
|
== builtin_info["output_cost_per_token_above_200k_tokens"]
|
|
)
|
|
|
|
|
|
def test_inherit_builtin_base_rates_for_off_peak_carries_companion_billing_fields():
|
|
"""Billing rules that are not literal cost rates, like the web search
|
|
billing unit, must ride along, or grounding and regional uplifts would
|
|
bill differently through the deployment entry than through the shared
|
|
backend entry.
|
|
"""
|
|
backend_model = "gemini-3-pro-image"
|
|
raw_entry = litellm.model_cost[backend_model]
|
|
assert raw_entry.get("web_search_billing_unit") is not None
|
|
|
|
model_info = {
|
|
"off_peak_pricing": {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-07},
|
|
}
|
|
|
|
Router._inherit_builtin_base_rates_for_off_peak(
|
|
model_info=model_info,
|
|
backend_model=backend_model,
|
|
custom_llm_provider=None,
|
|
)
|
|
|
|
assert model_info["web_search_billing_unit"] == raw_entry["web_search_billing_unit"]
|
|
assert model_info["input_cost_per_token"] == raw_entry["input_cost_per_token"]
|
|
|
|
|
|
def test_inherit_builtin_base_rates_for_off_peak_tiered_only_backend_stores_no_zero():
|
|
"""A tiered-only backend has no flat token rates; get_model_info synthesizes
|
|
zeros for them, and storing those would mark the deployment explicitly
|
|
priced free. The tier table itself must carry over as an isolated copy so
|
|
mutating the deployment entry never touches the shared cost map.
|
|
"""
|
|
backend_model = "dashscope/qwen-flash"
|
|
raw_tiers = litellm.model_cost[backend_model]["tiered_pricing"]
|
|
|
|
model_info = {
|
|
"off_peak_pricing": {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-07},
|
|
}
|
|
|
|
Router._inherit_builtin_base_rates_for_off_peak(
|
|
model_info=model_info,
|
|
backend_model=backend_model,
|
|
custom_llm_provider="dashscope",
|
|
)
|
|
|
|
assert model_info.get("input_cost_per_token") != 0
|
|
assert model_info.get("output_cost_per_token") != 0
|
|
assert model_info["tiered_pricing"] == raw_tiers
|
|
assert model_info["tiered_pricing"] is not raw_tiers
|
|
assert model_info["tiered_pricing"][0] is not raw_tiers[0]
|
|
|
|
original_first_tier = copy.deepcopy(raw_tiers[0])
|
|
model_info["tiered_pricing"][0]["input_cost_per_token"] = 123.0
|
|
assert raw_tiers[0] == original_first_tier
|
|
|
|
|
|
def test_inherit_builtin_base_rates_for_off_peak_leaves_explicit_rates_alone():
|
|
"""An entry that sets its own base rate beside the block already counts as
|
|
a full custom pricing entry; the helper must not mix builtin rates into it.
|
|
"""
|
|
model_info = {
|
|
"off_peak_pricing": {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-07},
|
|
"input_cost_per_token": 3e-06,
|
|
}
|
|
|
|
Router._inherit_builtin_base_rates_for_off_peak(
|
|
model_info=model_info,
|
|
backend_model="gpt-4o-mini",
|
|
custom_llm_provider="openai",
|
|
)
|
|
|
|
assert model_info["input_cost_per_token"] == 3e-06
|
|
assert "output_cost_per_token" not in model_info
|
|
|
|
|
|
def test_inherit_builtin_base_rates_for_off_peak_noop_without_block_or_backend():
|
|
"""Nothing happens without an off_peak_pricing block, and an unmapped
|
|
backend model leaves the entry unchanged rather than raising.
|
|
"""
|
|
plain_info = {"id": "dep-1"}
|
|
Router._inherit_builtin_base_rates_for_off_peak(
|
|
model_info=plain_info,
|
|
backend_model="gpt-4o-mini",
|
|
custom_llm_provider="openai",
|
|
)
|
|
assert plain_info == {"id": "dep-1"}
|
|
|
|
off_peak_info = {"off_peak_pricing": {"hours_utc": "00:00-00:00", "input_cost_per_token": 5e-07}}
|
|
Router._inherit_builtin_base_rates_for_off_peak(
|
|
model_info=off_peak_info,
|
|
backend_model="this-backend-model-does-not-exist-x9y8z7",
|
|
custom_llm_provider=None,
|
|
)
|
|
assert "input_cost_per_token" not in off_peak_info
|
|
|
|
|
|
def test_custom_pricing_field_denylist_covers_all_builtin_pricing_fields():
|
|
"""The shared-backend-key stripping in Router relies on
|
|
CustomPricingLiteLLMParams enumerating every per-deployment pricing field.
|
|
If a new pricing field is added to ModelInfoBase but not mirrored here, a
|
|
deployment override on that field leaks into the shared backend key and
|
|
every sibling deployment reads the wrong rate (LIT-3897). This guard fails
|
|
fast when the two drift apart.
|
|
"""
|
|
import typing
|
|
|
|
from litellm.types.utils import CustomPricingLiteLLMParams, ModelInfoBase
|
|
|
|
pricing_markers = ("cost", "price", "uplift", "vector_size", "tiered_pricing")
|
|
builtin_pricing_fields = {
|
|
name for name in typing.get_type_hints(ModelInfoBase) if any(marker in name for marker in pricing_markers)
|
|
}
|
|
denylisted_fields = set(CustomPricingLiteLLMParams.model_fields.keys())
|
|
|
|
uncovered = sorted(builtin_pricing_fields - denylisted_fields)
|
|
assert not uncovered, (
|
|
"ModelInfoBase pricing fields missing from CustomPricingLiteLLMParams; "
|
|
f"these would leak into shared backend keys: {uncovered}"
|
|
)
|
|
|
|
|
|
def test_tiered_pricing_override_isolated_from_sibling_via_model_info_lookup():
|
|
"""LIT-3897: a deployment that overrides a tiered pricing field
|
|
(input_cost_per_token_above_272k_tokens) must not pollute the shared
|
|
backend key, so a sibling sharing the same backend resolves its pricing
|
|
via litellm.get_model_info (the path /model/info uses) without seeing the
|
|
override.
|
|
"""
|
|
backend_model = "gemini/gemini-2.5-flash"
|
|
override = 0.000999
|
|
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
assert builtin_info.get("input_cost_per_token_above_272k_tokens") != override
|
|
|
|
model_keys = {
|
|
"lit3897-tiered-custom": litellm.model_cost.get("lit3897-tiered-custom"),
|
|
"lit3897-tiered-sibling": litellm.model_cost.get("lit3897-tiered-sibling"),
|
|
backend_model: copy.deepcopy(litellm.model_cost.get(backend_model)),
|
|
}
|
|
try:
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "custom-priced-flash",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-tiered-1",
|
|
},
|
|
"model_info": {
|
|
"id": "lit3897-tiered-custom",
|
|
"input_cost_per_token_above_272k_tokens": override,
|
|
"cache_read_input_token_cost_above_272k_tokens": override,
|
|
},
|
|
},
|
|
{
|
|
"model_name": "gemini-2.5-flash",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-tiered-2",
|
|
},
|
|
"model_info": {"id": "lit3897-tiered-sibling"},
|
|
},
|
|
],
|
|
)
|
|
|
|
shared = litellm.get_model_info(model=backend_model)
|
|
assert shared.get("input_cost_per_token_above_272k_tokens") != override, (
|
|
"Tiered override leaked into the shared backend key; siblings read the wrong rate via /model/info"
|
|
)
|
|
assert shared.get("cache_read_input_token_cost_above_272k_tokens") != override
|
|
|
|
custom_entry = litellm.model_cost["lit3897-tiered-custom"]
|
|
assert custom_entry["input_cost_per_token_above_272k_tokens"] == override
|
|
assert custom_entry["cache_read_input_token_cost_above_272k_tokens"] == override
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_custom_pricing_isolated_from_sibling_via_proxy_model_info_path():
|
|
"""LIT-3897 end to end through the proxy resolution helper: the override
|
|
deployment reports its custom input rate while the sibling keeps the
|
|
canonical gemini rate when /model/info resolves each deployment. Mirrors the
|
|
ticket config where the override is set on litellm_params.
|
|
"""
|
|
from litellm.proxy.proxy_server import _get_proxy_model_info
|
|
|
|
backend_model = "gemini/gemini-2.5-flash"
|
|
override_input = 5e-05
|
|
override_output = 1e-04
|
|
|
|
builtin_info = litellm.get_model_info(model=backend_model)
|
|
builtin_input = builtin_info["input_cost_per_token"]
|
|
assert builtin_input != override_input
|
|
|
|
model_keys = {
|
|
"lit3897-proxy-custom": litellm.model_cost.get("lit3897-proxy-custom"),
|
|
"lit3897-proxy-sibling": litellm.model_cost.get("lit3897-proxy-sibling"),
|
|
backend_model: copy.deepcopy(litellm.model_cost.get(backend_model)),
|
|
}
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "custom-priced-flash",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-proxy-1",
|
|
"input_cost_per_token": override_input,
|
|
"output_cost_per_token": override_output,
|
|
},
|
|
"model_info": {"id": "lit3897-proxy-custom"},
|
|
},
|
|
{
|
|
"model_name": "gemini-2.5-flash",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-proxy-2",
|
|
},
|
|
"model_info": {"id": "lit3897-proxy-sibling"},
|
|
},
|
|
],
|
|
)
|
|
|
|
resolved = {
|
|
m["model_name"]: _get_proxy_model_info(model=copy.deepcopy(m))["model_info"]["input_cost_per_token"]
|
|
for m in router.model_list
|
|
}
|
|
|
|
assert resolved["custom-priced-flash"] == override_input
|
|
assert resolved["gemini-2.5-flash"] == builtin_input
|
|
assert resolved["gemini-2.5-flash"] != resolved["custom-priced-flash"]
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_custom_model_info_metadata_not_leaked_to_shared_backend_key():
|
|
"""LIT-4544: two deployments share the same backend model but carry
|
|
different custom model_info (arbitrary keys, access_via_team_ids, ids).
|
|
None of that per-deployment metadata may land on the shared backend key in
|
|
litellm.model_cost (served raw by /public/litellm_model_cost_map);
|
|
before the fix it was merged last-write-wins so values flipped randomly.
|
|
"""
|
|
backend_model = "openai/gpt-4o-mini"
|
|
shared_keys = ("gpt-4o-mini", backend_model)
|
|
leak_fields = ("id", "additionalProp1", "access_via_team_ids", "db_model")
|
|
|
|
model_keys = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (*shared_keys, "lit4544-deploy-a", "lit4544-deploy-b")
|
|
}
|
|
try:
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "alias-unrestricted",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-a",
|
|
},
|
|
"model_info": {
|
|
"id": "lit4544-deploy-a",
|
|
"additionalProp1": {"restricted": False, "model_location": "EU"},
|
|
},
|
|
},
|
|
{
|
|
"model_name": "alias-restricted",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key-b",
|
|
},
|
|
"model_info": {
|
|
"id": "lit4544-deploy-b",
|
|
"additionalProp1": {"restricted": True, "model_location": "US"},
|
|
"access_via_team_ids": ["team-b-only"],
|
|
},
|
|
},
|
|
],
|
|
)
|
|
|
|
for shared_key in shared_keys:
|
|
shared_entry = litellm.model_cost.get(shared_key) or {}
|
|
leaked = [field for field in leak_fields if field in shared_entry]
|
|
assert not leaked, f"per-deployment metadata {leaked} leaked onto shared key {shared_key}: {shared_entry}"
|
|
|
|
entry_a = litellm.model_cost["lit4544-deploy-a"]
|
|
assert entry_a["additionalProp1"] == {"restricted": False, "model_location": "EU"}
|
|
entry_b = litellm.model_cost["lit4544-deploy-b"]
|
|
assert entry_b["additionalProp1"] == {"restricted": True, "model_location": "US"}
|
|
assert entry_b["access_via_team_ids"] == ["team-b-only"]
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_add_deployment_does_not_leak_custom_metadata_to_shared_backend_key():
|
|
"""LIT-4544 dynamic path: deployments added at runtime (e.g. loaded from
|
|
the DB every scheduler cycle) must not re-pollute the shared backend key
|
|
with per-deployment metadata either.
|
|
"""
|
|
backend_model = "openai/gpt-4o-mini"
|
|
shared_keys = ("gpt-4o-mini", backend_model)
|
|
deploy_id = "lit4544-add-deployment"
|
|
|
|
model_keys = {key: copy.deepcopy(litellm.model_cost.get(key)) for key in (*shared_keys, deploy_id)}
|
|
try:
|
|
router = Router(model_list=[])
|
|
router.add_deployment(
|
|
deployment=Deployment(
|
|
model_name="alias-dynamic",
|
|
litellm_params=LiteLLM_Params(
|
|
model=backend_model,
|
|
api_key="fake-key-dynamic",
|
|
),
|
|
model_info=ModelInfo(
|
|
id=deploy_id,
|
|
additionalProp1={"restricted": True},
|
|
access_via_team_ids=["team-dynamic"],
|
|
),
|
|
)
|
|
)
|
|
|
|
for shared_key in shared_keys:
|
|
shared_entry = litellm.model_cost.get(shared_key) or {}
|
|
leaked = [
|
|
field for field in ("id", "additionalProp1", "access_via_team_ids", "db_model") if field in shared_entry
|
|
]
|
|
assert not leaked, f"per-deployment metadata {leaked} leaked onto shared key {shared_key}: {shared_entry}"
|
|
|
|
assert litellm.model_cost[deploy_id]["access_via_team_ids"] == ["team-dynamic"]
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_shared_backend_model_info_keeps_schema_fields_and_drops_the_rest():
|
|
"""Unit test of the whitelist helper: cost-map schema fields survive,
|
|
custom pricing overrides and per-deployment metadata do not.
|
|
"""
|
|
from litellm.types.utils import shared_backend_model_info
|
|
|
|
filtered = shared_backend_model_info(
|
|
{
|
|
"mode": "chat",
|
|
"litellm_provider": "openai",
|
|
"max_tokens": 128000,
|
|
"supports_vision": True,
|
|
"supported_endpoints": ["/v1/responses"],
|
|
"use_openai_responses_path": True,
|
|
"input_cost_per_token": 0.99,
|
|
"output_cost_per_token": 0.99,
|
|
"id": "deploy-a",
|
|
"db_model": False,
|
|
"access_via_team_ids": ["team-a"],
|
|
"additionalProp1": {"restricted": True},
|
|
"base_model": "gpt-4o-mini",
|
|
}
|
|
)
|
|
|
|
assert filtered == {
|
|
"mode": "chat",
|
|
"litellm_provider": "openai",
|
|
"max_tokens": 128000,
|
|
"supports_vision": True,
|
|
"supported_endpoints": ["/v1/responses"],
|
|
"use_openai_responses_path": True,
|
|
}
|
|
|
|
|
|
def test_capability_flags_propagate_from_deployment_model_info_to_shared_key():
|
|
"""Backend-model capability facts (supported_endpoints,
|
|
use_openai_responses_path) declared in a deployment's model_info must reach
|
|
the shared backend key: the Bedrock Mantle routing gates read them raw off
|
|
litellm.model_cost and document proxy model_info as an override path for
|
|
models missing from the built-in cost map.
|
|
"""
|
|
from litellm.llms.bedrock_mantle.common_utils import (
|
|
mantle_base_segment,
|
|
mantle_supports_responses,
|
|
)
|
|
|
|
bare_model = "somelab.lit4544-unmapped-model"
|
|
backend_model = f"bedrock_mantle/{bare_model}"
|
|
deploy_id = "lit4544-mantle-deploy"
|
|
|
|
model_keys = {key: copy.deepcopy(litellm.model_cost.get(key)) for key in (bare_model, backend_model, deploy_id)}
|
|
try:
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "mantle-alias",
|
|
"litellm_params": {
|
|
"model": backend_model,
|
|
"api_key": "fake-key",
|
|
},
|
|
"model_info": {
|
|
"id": deploy_id,
|
|
"supported_endpoints": ["/v1/responses"],
|
|
"use_openai_responses_path": True,
|
|
},
|
|
},
|
|
],
|
|
)
|
|
|
|
shared_entry = litellm.model_cost.get(backend_model) or {}
|
|
assert shared_entry.get("supported_endpoints") == ["/v1/responses"]
|
|
assert shared_entry.get("use_openai_responses_path") is True
|
|
assert "id" not in shared_entry
|
|
assert mantle_supports_responses(bare_model, litellm.model_cost) is True
|
|
assert mantle_base_segment(bare_model, litellm.model_cost) == "openai/v1"
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_wildcard_zero_cost_request_does_not_poison_named_deployment_pricing():
|
|
"""LIT-3991 end to end: a proxy has a named text-embedding-3-small
|
|
deployment relying on built-in pricing plus an ``openai/*`` wildcard with
|
|
explicit zero pricing. One embedding call routed through the wildcard must
|
|
not clobber the shared ``openai/text-embedding-3-small`` pricing; requests
|
|
to the named deployment afterwards must still cost non-zero.
|
|
"""
|
|
shared_key = "openai/text-embedding-3-small"
|
|
model_keys = {
|
|
shared_key: copy.deepcopy(litellm.model_cost.get(shared_key)),
|
|
"text-embedding-3-small": copy.deepcopy(litellm.model_cost.get("text-embedding-3-small")),
|
|
"openai/*": copy.deepcopy(litellm.model_cost.get("openai/*")),
|
|
"lit3991-named": litellm.model_cost.get("lit3991-named"),
|
|
"lit3991-wildcard": litellm.model_cost.get("lit3991-wildcard"),
|
|
}
|
|
builtin_input_cost = litellm.get_model_info(model=shared_key)["input_cost_per_token"]
|
|
assert builtin_input_cost > 0
|
|
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "text-embedding-3-small",
|
|
"litellm_params": {
|
|
"model": "openai/text-embedding-3-small",
|
|
"api_key": "fake-key-named",
|
|
},
|
|
"model_info": {"id": "lit3991-named"},
|
|
},
|
|
{
|
|
"model_name": "openai/*",
|
|
"litellm_params": {
|
|
"model": "openai/*",
|
|
"api_key": "fake-key-wildcard",
|
|
"input_cost_per_token": 0.0,
|
|
"output_cost_per_token": 0.0,
|
|
},
|
|
"model_info": {"id": "lit3991-wildcard"},
|
|
},
|
|
],
|
|
)
|
|
|
|
router.embedding(
|
|
model="openai/text-embedding-3-small",
|
|
input=["hello"],
|
|
mock_response=[0.1, 0.2],
|
|
)
|
|
|
|
assert litellm.get_model_info(model=shared_key)["input_cost_per_token"] == builtin_input_cost, (
|
|
f"one call through the zero-cost wildcard poisoned the shared {shared_key} pricing for the named deployment"
|
|
)
|
|
|
|
named_response = router.embedding(
|
|
model="text-embedding-3-small",
|
|
input=["hello"],
|
|
mock_response=[0.1, 0.2],
|
|
)
|
|
named_cost = litellm.completion_cost(completion_response=named_response, call_type="embedding")
|
|
assert named_cost == pytest.approx(10 * builtin_input_cost)
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_price_data_reload_preserves_router_registered_model_info(monkeypatch):
|
|
"""
|
|
A price-data reload replaces litellm.model_cost wholesale. Deployment
|
|
model_info registered by the Router is not in the fetched catalog, so
|
|
without a replay of runtime registrations the reload silently strips
|
|
max_input_tokens / max_output_tokens from every custom model group and
|
|
/model_group/info starts reporting nulls.
|
|
"""
|
|
from litellm import utils as litellm_utils
|
|
|
|
monkeypatch.setattr(
|
|
litellm_utils,
|
|
"_runtime_registered_model_cost",
|
|
dict(litellm_utils._runtime_registered_model_cost),
|
|
)
|
|
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "custom-alias",
|
|
"litellm_params": {"model": "hosted_vllm/not-in-the-catalog"},
|
|
"model_info": {
|
|
"id": "custom-alias-id",
|
|
"max_input_tokens": 128000,
|
|
"max_output_tokens": 16384,
|
|
},
|
|
}
|
|
],
|
|
)
|
|
|
|
before = router.get_model_group_info(model_group="custom-alias")
|
|
assert before is not None
|
|
assert before.max_input_tokens == 128000
|
|
assert before.max_output_tokens == 16384
|
|
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
_simulate_price_data_reload(
|
|
{"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}},
|
|
)
|
|
|
|
after = router.get_model_group_info(model_group="custom-alias")
|
|
assert after is not None
|
|
assert after.max_input_tokens == 128000
|
|
assert after.max_output_tokens == 16384
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_price_data_reload_preserves_custom_override_of_a_catalog_model(monkeypatch):
|
|
"""
|
|
A deployment whose backend model IS in the catalog is the quieter half of
|
|
the same bug: the reload does not blank the metadata, it reverts the
|
|
operator's model_info override to the upstream catalog values.
|
|
"""
|
|
from litellm import utils as litellm_utils
|
|
|
|
monkeypatch.setattr(
|
|
litellm_utils,
|
|
"_runtime_registered_model_cost",
|
|
dict(litellm_utils._runtime_registered_model_cost),
|
|
)
|
|
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "capped-gpt-4o",
|
|
"litellm_params": {"model": "openai/gpt-4o"},
|
|
"model_info": {
|
|
"id": "capped-gpt-4o-id",
|
|
"max_input_tokens": 12345,
|
|
"max_output_tokens": 678,
|
|
},
|
|
}
|
|
],
|
|
)
|
|
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
_simulate_price_data_reload(
|
|
{
|
|
"openai/gpt-4o": {
|
|
"litellm_provider": "openai",
|
|
"mode": "chat",
|
|
"max_input_tokens": 999999,
|
|
"max_output_tokens": 888888,
|
|
}
|
|
},
|
|
)
|
|
|
|
after = router.get_model_group_info(model_group="capped-gpt-4o")
|
|
assert after is not None
|
|
assert after.max_input_tokens == 12345
|
|
assert after.max_output_tokens == 678
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_deleted_deployments_are_not_replayed_onto_later_reloads(monkeypatch):
|
|
"""
|
|
Runtime registrations are replayed onto every price data reload, so a
|
|
deleted deployment has to be withdrawn or it is re-asserted for the life of
|
|
the process and the registry grows with every create/delete cycle. A backend
|
|
key that another live deployment still points at must survive the same
|
|
deletion.
|
|
"""
|
|
from litellm import utils as litellm_utils
|
|
|
|
monkeypatch.setattr(
|
|
litellm_utils,
|
|
"_runtime_registered_model_cost",
|
|
dict(litellm_utils._runtime_registered_model_cost),
|
|
)
|
|
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "doomed",
|
|
"litellm_params": {"model": "hosted_vllm/shared-backend"},
|
|
"model_info": {"id": "doomed-id", "max_input_tokens": 111},
|
|
},
|
|
{
|
|
"model_name": "kept",
|
|
"litellm_params": {"model": "hosted_vllm/shared-backend"},
|
|
"model_info": {"id": "kept-id", "max_input_tokens": 222},
|
|
},
|
|
{
|
|
"model_name": "solo",
|
|
"litellm_params": {"model": "hosted_vllm/solo-backend"},
|
|
"model_info": {"id": "solo-id", "max_input_tokens": 333},
|
|
},
|
|
],
|
|
)
|
|
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
assert router.delete_deployment(id="doomed-id") is not None
|
|
assert router.delete_deployment(id="solo-id") is not None
|
|
|
|
_simulate_price_data_reload(
|
|
{"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}},
|
|
)
|
|
|
|
assert "doomed-id" not in litellm.model_cost
|
|
assert "solo-id" not in litellm.model_cost
|
|
assert "hosted_vllm/solo-backend" not in litellm.model_cost
|
|
|
|
surviving = litellm.model_cost["kept-id"]
|
|
assert surviving["max_input_tokens"] == 222
|
|
assert "hosted_vllm/shared-backend" in litellm.model_cost
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_deleting_a_deployment_leaves_catalog_pricing_for_its_backend_model(monkeypatch):
|
|
"""
|
|
A backend key is shared with the fetched catalog, so withdrawing the entries
|
|
a deleted deployment owns must not take real upstream pricing down with it.
|
|
"""
|
|
from litellm import utils as litellm_utils
|
|
|
|
monkeypatch.setattr(
|
|
litellm_utils,
|
|
"_runtime_registered_model_cost",
|
|
dict(litellm_utils._runtime_registered_model_cost),
|
|
)
|
|
|
|
backend_model = "gemini/gemini-2.5-pro"
|
|
catalog_entry = litellm.get_model_info(model=backend_model)
|
|
catalog_input_cost = catalog_entry["input_cost_per_token"]
|
|
assert catalog_input_cost > 0, "Test requires a catalog model with non-zero pricing"
|
|
|
|
saved_catalog = litellm.model_cost
|
|
fetched_catalog = copy.deepcopy(litellm.model_cost)
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "doomed-gemini",
|
|
"litellm_params": {"model": backend_model, "api_key": "sk-fake"},
|
|
"model_info": {"id": "doomed-gemini-id"},
|
|
}
|
|
],
|
|
)
|
|
|
|
assert router.delete_deployment(id="doomed-gemini-id") is not None
|
|
|
|
_simulate_price_data_reload(
|
|
copy.deepcopy(fetched_catalog),
|
|
)
|
|
|
|
assert "doomed-gemini-id" not in litellm.model_cost
|
|
assert litellm.model_cost[backend_model]["input_cost_per_token"] == catalog_input_cost
|
|
finally:
|
|
litellm.model_cost = saved_catalog
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_repointing_a_deployment_drops_its_previous_backend_key(monkeypatch):
|
|
"""
|
|
An update that moves a deployment onto a different backend model leaves the
|
|
old backend key behind, and a replayed registry would re-assert it onto every
|
|
later catalog for the life of the process.
|
|
"""
|
|
from litellm import utils as litellm_utils
|
|
|
|
monkeypatch.setattr(
|
|
litellm_utils,
|
|
"_runtime_registered_model_cost",
|
|
dict(litellm_utils._runtime_registered_model_cost),
|
|
)
|
|
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "moving-target",
|
|
"litellm_params": {"model": "hosted_vllm/old-backend"},
|
|
"model_info": {"id": "moving-target-id"},
|
|
}
|
|
],
|
|
)
|
|
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
router.upsert_deployment(
|
|
deployment=Deployment(
|
|
model_name="moving-target",
|
|
litellm_params=LiteLLM_Params(model="hosted_vllm/new-backend"),
|
|
model_info=ModelInfo(id="moving-target-id"),
|
|
)
|
|
)
|
|
|
|
_simulate_price_data_reload(
|
|
{"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}},
|
|
)
|
|
|
|
assert "hosted_vllm/old-backend" not in litellm.model_cost
|
|
assert "hosted_vllm/new-backend" in litellm.model_cost
|
|
assert "moving-target-id" in litellm.model_cost
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"model, custom_llm_provider, expected",
|
|
[
|
|
("gpt-4o", None, ("gpt-4o",)),
|
|
("gpt-4o", "openai", ("openai/gpt-4o",)),
|
|
("openai/gpt-4o", None, ("openai/gpt-4o",)),
|
|
("responses/gpt-4o", "openai", ("openai/responses/gpt-4o", "openai/gpt-4o")),
|
|
("responses/gpt-4o", None, ("responses/gpt-4o", "gpt-4o")),
|
|
],
|
|
)
|
|
def test_backend_cost_map_keys_matches_what_registration_writes(model, custom_llm_provider, expected):
|
|
"""
|
|
The withdrawal path drops exactly the keys the registration wrote, so the two
|
|
have to agree on the provider prefix and on the responses/ alias. The first
|
|
key is also the one the registration uses as the shared backend key, so its
|
|
position is load-bearing rather than incidental.
|
|
"""
|
|
keys = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider)
|
|
assert keys == expected
|
|
assert keys[0] == (model if custom_llm_provider is None else f"{custom_llm_provider}/{model}")
|
|
|
|
|
|
def test_a_discarded_router_stops_contributing_to_later_reloads(monkeypatch):
|
|
"""
|
|
`_route_user_config_request` builds a Router per request from caller-supplied
|
|
config and discards it. Nothing can withdraw entries on its behalf afterwards,
|
|
so a rebuild driven off live routers is what keeps a caller from growing the
|
|
cost map one request at a time.
|
|
"""
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
kept = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "kept",
|
|
"litellm_params": {"model": "hosted_vllm/kept-backend"},
|
|
"model_info": {"id": "kept-router-id", "max_input_tokens": 4242},
|
|
}
|
|
],
|
|
)
|
|
throwaway = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "throwaway",
|
|
"litellm_params": {"model": "hosted_vllm/throwaway-backend"},
|
|
"model_info": {"id": "throwaway-router-id", "max_input_tokens": 111},
|
|
}
|
|
],
|
|
)
|
|
throwaway.discard()
|
|
|
|
_simulate_price_data_reload(
|
|
{"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}},
|
|
)
|
|
|
|
assert "throwaway-router-id" not in litellm.model_cost
|
|
assert "hosted_vllm/throwaway-backend" not in litellm.model_cost
|
|
assert litellm.model_cost["kept-router-id"]["max_input_tokens"] == 4242
|
|
assert "hosted_vllm/kept-backend" in litellm.model_cost
|
|
assert kept.model_list # keep the live router referenced for the duration
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_a_reload_rebuilds_exactly_what_a_fresh_boot_registered() -> None:
|
|
"""
|
|
The rebuild is only correct if it reproduces the entries the original
|
|
registration wrote, including the pieces that are derived rather than stored:
|
|
custom pricing carried on litellm_params, and the cache pricing inherited from
|
|
the built-in cost map.
|
|
"""
|
|
saved_catalog = litellm.model_cost
|
|
fetched_catalog = copy.deepcopy(litellm.model_cost)
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "priced",
|
|
"litellm_params": {
|
|
"model": "openai/gpt-4o",
|
|
"api_key": "sk-fake",
|
|
"input_cost_per_token": 0.000123,
|
|
"output_cost_per_token": 0.000456,
|
|
},
|
|
"model_info": {"id": "priced-id", "max_input_tokens": 4242},
|
|
}
|
|
],
|
|
)
|
|
at_boot = copy.deepcopy(litellm.model_cost["priced-id"])
|
|
assert at_boot["input_cost_per_token"] == 0.000123
|
|
assert at_boot["cache_read_input_token_cost"] is not None
|
|
assert "member_auto_router" not in litellm.model_cost["gpt-4o"]
|
|
|
|
_simulate_price_data_reload(
|
|
copy.deepcopy(fetched_catalog),
|
|
)
|
|
|
|
rebuilt = litellm.model_cost["priced-id"]
|
|
assert at_boot.items() <= rebuilt.items(), (
|
|
f"the rebuild changed or dropped a field the boot registration wrote: "
|
|
f"{ {k: (v, rebuilt.get(k)) for k, v in at_boot.items() if rebuilt.get(k) != v} }"
|
|
)
|
|
assert {field: rebuilt[field] for field in set(rebuilt) - set(at_boot)} == {
|
|
"db_model": False,
|
|
"member_auto_router": False,
|
|
}
|
|
assert "member_auto_router" not in litellm.model_cost["gpt-4o"]
|
|
assert router.model_list
|
|
finally:
|
|
litellm.model_cost = saved_catalog
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_replay_model_cost_registrations_survives_a_malformed_deployment():
|
|
"""
|
|
The rebuild reads whatever dicts are sitting in model_list, so one entry that
|
|
cannot be rebuilt into a Deployment must not stop the rest being restored.
|
|
"""
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "healthy",
|
|
"litellm_params": {"model": "hosted_vllm/healthy-backend"},
|
|
"model_info": {"id": "healthy-id", "max_input_tokens": 777},
|
|
}
|
|
],
|
|
)
|
|
router.model_list.insert(0, {"litellm_params": {}})
|
|
|
|
litellm.model_cost = {"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}}
|
|
_invalidate_model_cost_lowercase_map()
|
|
router._replay_model_cost_registrations()
|
|
|
|
assert litellm.model_cost["healthy-id"]["max_input_tokens"] == 777
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_deployment_model_cost_payload_folds_in_litellm_params_pricing():
|
|
"""
|
|
Custom pricing is configured on litellm_params but has to land in the
|
|
cost-map entry, and setting it pulls in the built-in cache pricing for the
|
|
backend model. Both are what make the entry reproducible from a deployment.
|
|
"""
|
|
payload = Router._deployment_model_cost_payload(
|
|
deployment=Deployment(
|
|
model_name="priced",
|
|
litellm_params=LiteLLM_Params(
|
|
model="gemini/gemini-2.5-pro",
|
|
input_cost_per_token=0.000123,
|
|
),
|
|
model_info=ModelInfo(id="payload-id", max_input_tokens=4242),
|
|
)
|
|
)
|
|
|
|
assert payload["id"] == "payload-id"
|
|
assert payload["max_input_tokens"] == 4242
|
|
assert payload["input_cost_per_token"] == 0.000123
|
|
assert payload["cache_read_input_token_cost"] > 0
|
|
|
|
|
|
def test_deployment_model_cost_payload_includes_builtin_service_tier_pricing() -> None:
|
|
model_id: Final = "tier-priced-payload"
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key))
|
|
for key in (_TIER_BACKEND_KEY, _TIER_BACKEND_MODEL, model_id)
|
|
}
|
|
try:
|
|
_register_tier_backend()
|
|
payload: Final = Router._deployment_model_cost_payload(
|
|
deployment=Deployment(
|
|
model_name="tier-priced-payload",
|
|
litellm_params=LiteLLM_Params(
|
|
model=_TIER_BACKEND_MODEL,
|
|
custom_llm_provider="openai",
|
|
input_cost_per_token=_CUSTOM_STANDARD_INPUT_RATE,
|
|
output_cost_per_token=_CUSTOM_STANDARD_OUTPUT_RATE,
|
|
),
|
|
model_info=ModelInfo(id=model_id),
|
|
)
|
|
)
|
|
|
|
assert (
|
|
payload["input_cost_per_token_ultrafast"] == _TIER_BACKEND_ENTRY["input_cost_per_token_ultrafast"]
|
|
)
|
|
assert (
|
|
payload["output_cost_per_token_ultrafast"] == _TIER_BACKEND_ENTRY["output_cost_per_token_ultrafast"]
|
|
)
|
|
assert payload["input_cost_per_token_balanced"] == _TIER_BACKEND_ENTRY["input_cost_per_token_balanced"]
|
|
assert payload["input_cost_per_token"] == _CUSTOM_STANDARD_INPUT_RATE
|
|
assert payload["output_cost_per_token"] == _CUSTOM_STANDARD_OUTPUT_RATE
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_register_deployment_in_model_cost_writes_both_key_families():
|
|
"""
|
|
A deployment contributes its full model_info under its unique id and the
|
|
cost-map subset under the shared backend key, and the shared key must not
|
|
pick up the deployment's private metadata.
|
|
"""
|
|
model_keys = {
|
|
"both-families-id": copy.deepcopy(litellm.model_cost.get("both-families-id")),
|
|
"hosted_vllm/both-families-backend": copy.deepcopy(litellm.model_cost.get("hosted_vllm/both-families-backend")),
|
|
}
|
|
try:
|
|
Router._register_deployment_in_model_cost(
|
|
model_id="both-families-id",
|
|
model_info={"id": "both-families-id", "max_input_tokens": 999, "litellm_provider": "hosted_vllm"},
|
|
model="hosted_vllm/both-families-backend",
|
|
custom_llm_provider=None,
|
|
)
|
|
|
|
assert litellm.model_cost["both-families-id"]["max_input_tokens"] == 999
|
|
shared = litellm.model_cost["hosted_vllm/both-families-backend"]
|
|
assert shared["max_input_tokens"] == 999
|
|
assert "id" not in shared
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
|
|
|
|
def test_router_registration_keeps_ultrafast_long_context_deployment_pricing() -> None:
|
|
model_id: Final = "ultrafast-long-context-pricing-id"
|
|
backend_key: Final = "openai/gpt-6-astra"
|
|
rates: Final = {
|
|
"input_cost_per_token_above_272k_tokens_ultrafast": 0.00012,
|
|
"output_cost_per_token_above_272k_tokens_ultrafast": 0.00045,
|
|
"cache_read_input_token_cost_above_272k_tokens_ultrafast": 1.2e-05,
|
|
"cache_creation_input_token_cost_above_272k_tokens_ultrafast": 0.00015,
|
|
}
|
|
model_cost_entries: Final = {
|
|
key: copy.deepcopy(litellm.model_cost.get(key)) for key in (model_id, backend_key, "gpt-6-astra")
|
|
}
|
|
try:
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "ultrafast-long-context-pricing",
|
|
"litellm_params": {"model": backend_key, **rates},
|
|
"model_info": {"id": model_id},
|
|
}
|
|
]
|
|
)
|
|
|
|
assert {key: litellm.model_cost[model_id][key] for key in rates} == rates
|
|
finally:
|
|
_restore_model_cost_entries(model_cost_entries)
|
|
|
|
|
|
def test_reload_keeps_custom_pricing_configured_on_litellm_params_for_a_db_model():
|
|
"""
|
|
A deployment added at runtime, which is what /model/new does, configures its
|
|
custom pricing on litellm_params rather than on model_info. A price data
|
|
reload must not revert that to the catalog's pricing.
|
|
"""
|
|
saved_catalog = litellm.model_cost
|
|
fetched_catalog = copy.deepcopy(litellm.model_cost)
|
|
try:
|
|
router = Router(model_list=[])
|
|
router.add_deployment(
|
|
deployment=Deployment(
|
|
model_name="db-priced",
|
|
litellm_params=LiteLLM_Params(
|
|
model="openai/gpt-4o",
|
|
api_key="sk-fake",
|
|
input_cost_per_token=0.000123,
|
|
output_cost_per_token=0.000456,
|
|
),
|
|
model_info=ModelInfo(id="db-priced-id"),
|
|
)
|
|
)
|
|
|
|
assert litellm.model_cost["db-priced-id"]["input_cost_per_token"] == 0.000123
|
|
|
|
_simulate_price_data_reload(
|
|
copy.deepcopy(fetched_catalog),
|
|
)
|
|
|
|
assert litellm.model_cost["db-priced-id"]["input_cost_per_token"] == 0.000123
|
|
assert litellm.model_cost["db-priced-id"]["output_cost_per_token"] == 0.000456
|
|
finally:
|
|
litellm.model_cost = saved_catalog
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_replay_live_router_model_cost_rebuilds_every_live_router():
|
|
"""
|
|
A process can hold more than one Router, so the rebuild has to fan out across
|
|
all of them rather than restoring whichever one happens to be reachable.
|
|
"""
|
|
from litellm.router import _replay_live_router_model_cost
|
|
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
first = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "first",
|
|
"litellm_params": {"model": "hosted_vllm/first-backend"},
|
|
"model_info": {"id": "first-id", "max_input_tokens": 111},
|
|
}
|
|
],
|
|
)
|
|
second = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "second",
|
|
"litellm_params": {"model": "hosted_vllm/second-backend"},
|
|
"model_info": {"id": "second-id", "max_input_tokens": 222},
|
|
}
|
|
],
|
|
)
|
|
|
|
litellm.model_cost = {"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}}
|
|
_invalidate_model_cost_lowercase_map()
|
|
_replay_live_router_model_cost()
|
|
|
|
assert litellm.model_cost["first-id"]["max_input_tokens"] == 111
|
|
assert litellm.model_cost["second-id"]["max_input_tokens"] == 222
|
|
assert first.model_list and second.model_list
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_strategy_router_alias_pricing_never_enters_model_cost(monkeypatch):
|
|
"""
|
|
A strategy-router alias is never the deployment actually called or billed,
|
|
so custom pricing configured on it must not be registered under its
|
|
model_id - an explicit zero there makes the budget check treat the alias
|
|
as a genuinely free model while requests bill as a real deployment. The
|
|
strip must also survive a price-data reload, which rebuilds entries by
|
|
walking the live routers.
|
|
"""
|
|
from litellm import utils as litellm_utils
|
|
|
|
monkeypatch.setattr(
|
|
litellm_utils,
|
|
"_runtime_registered_model_cost",
|
|
dict(litellm_utils._runtime_registered_model_cost),
|
|
)
|
|
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "smart-router",
|
|
"litellm_params": {
|
|
"model": "auto_router/complexity_router/smart-router",
|
|
"complexity_router_default_model": "paid-model",
|
|
"input_cost_per_token": 0.0,
|
|
"output_cost_per_token": 0.0,
|
|
"complexity_router_config": {"tiers": {"simple": "paid-model"}},
|
|
},
|
|
"model_info": {"id": "strategy-alias-id", "max_input_tokens": 128000},
|
|
},
|
|
{
|
|
"model_name": "paid-model",
|
|
"litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"},
|
|
"model_info": {"id": "strategy-alias-paid-id"},
|
|
},
|
|
],
|
|
)
|
|
|
|
def _assert_alias_unpriced():
|
|
entry = litellm.model_cost.get("strategy-alias-id")
|
|
assert entry is not None, "Alias metadata should still be registered"
|
|
assert entry["max_input_tokens"] == 128000
|
|
assert "input_cost_per_token" not in entry
|
|
assert "output_cost_per_token" not in entry
|
|
|
|
_assert_alias_unpriced()
|
|
|
|
saved_model_cost = litellm.model_cost
|
|
try:
|
|
_simulate_price_data_reload(
|
|
{"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}},
|
|
)
|
|
_assert_alias_unpriced()
|
|
assert router.model_list
|
|
finally:
|
|
litellm.model_cost = saved_model_cost
|
|
_invalidate_model_cost_lowercase_map()
|
|
|
|
|
|
def test_inherit_builtin_tiered_output_rate_fills_the_backend_flat_rate():
|
|
"""
|
|
A deployment entry whose custom tiers publish only input rates would bill
|
|
completions at 0, so the backend model's flat output rate is copied in at
|
|
registration.
|
|
"""
|
|
model_info = {"tiered_pricing": [{"range": [0, 3000], "input_cost_per_token": 3.25e-07}]}
|
|
|
|
Router._inherit_builtin_tiered_output_rate(
|
|
model_info=model_info,
|
|
backend_model="claude-haiku-4-5",
|
|
custom_llm_provider="anthropic",
|
|
)
|
|
|
|
backend_rate = litellm.get_model_info(model="claude-haiku-4-5", custom_llm_provider="anthropic")[
|
|
"output_cost_per_token"
|
|
]
|
|
assert backend_rate > 0
|
|
assert model_info["output_cost_per_token"] == backend_rate
|
|
|
|
|
|
def test_inherit_builtin_tiered_output_rate_never_stores_a_synthesized_zero():
|
|
"""
|
|
Regression: get_model_info reports output_cost_per_token 0 for a backend that
|
|
only publishes tiered rates (e.g. dashscope/qwen-flash), and storing that zero
|
|
would mark the deployment as explicitly priced free.
|
|
"""
|
|
backend_info = litellm.get_model_info(model="qwen-flash", custom_llm_provider="dashscope")
|
|
assert backend_info["output_cost_per_token"] == 0
|
|
|
|
model_info = {"tiered_pricing": [{"range": [0, 3000], "input_cost_per_token": 3.25e-07}]}
|
|
Router._inherit_builtin_tiered_output_rate(
|
|
model_info=model_info,
|
|
backend_model="qwen-flash",
|
|
custom_llm_provider="dashscope",
|
|
)
|
|
|
|
assert "output_cost_per_token" not in model_info
|
|
|
|
|
|
def test_inherit_builtin_tiered_output_rate_leaves_a_user_rate_alone():
|
|
model_info = {
|
|
"tiered_pricing": [{"range": [0, 3000], "input_cost_per_token": 3.25e-07}],
|
|
"output_cost_per_token": 9e-07,
|
|
}
|
|
|
|
Router._inherit_builtin_tiered_output_rate(
|
|
model_info=model_info,
|
|
backend_model="claude-haiku-4-5",
|
|
custom_llm_provider="anthropic",
|
|
)
|
|
|
|
assert model_info["output_cost_per_token"] == 9e-07
|
|
|
|
|
|
# --- a config.yaml PTU deployment must not also bill per token ------------------
|
|
|
|
_PTU_MODEL_INFO = {
|
|
"id": "ptu-alpha-eastus",
|
|
"team_id": "team-alpha",
|
|
"ptu_count": 100,
|
|
"cost_per_ptu_per_hour": 0.02,
|
|
"ptu_effective_from": "2026-01-01T00:00:00Z",
|
|
}
|
|
|
|
|
|
def _ptu_router(model_info=None, litellm_params=None, ptu_enabled=True):
|
|
"""A router built the way loading config.yaml builds one."""
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": "True" if ptu_enabled else ""}, clear=False):
|
|
return Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "gpt-4o-ptu",
|
|
"litellm_params": {
|
|
"model": "anthropic/claude-sonnet-4-5-20250929",
|
|
"api_key": "sk-not-used",
|
|
**(litellm_params or {}),
|
|
},
|
|
"model_info": dict(_PTU_MODEL_INFO if model_info is None else model_info),
|
|
}
|
|
]
|
|
)
|
|
|
|
|
|
def test_a_config_ptu_deployment_bills_nothing_per_token():
|
|
"""Reserved capacity is already billed by the hour, so charging its traffic bills the
|
|
same tokens twice. Left unset the rate falls back to the public cost map, which makes
|
|
the double charge the default rather than an opt-in."""
|
|
router = _ptu_router(litellm_params={"input_cost_per_token": 5e-06, "output_cost_per_token": 1.5e-05})
|
|
entry = router.model_list[0]
|
|
|
|
assert entry["litellm_params"]["input_cost_per_token"] == 0.0
|
|
assert entry["litellm_params"]["output_cost_per_token"] == 0.0
|
|
assert entry["model_info"]["input_cost_per_token"] == 0.0
|
|
assert litellm.model_cost[entry["model_info"]["id"]]["input_cost_per_token"] == 0.0
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"backend",
|
|
["anthropic/claude-sonnet-4-5-20250929", "azure/gpt-4o", "gemini/gemini-2.5-flash"],
|
|
)
|
|
def test_a_config_ptu_deployment_imports_no_cache_rate_from_its_backend(backend):
|
|
"""The cache back-fill runs whenever input_cost_per_token is set, and 0.0 is set, so a
|
|
partially zeroed deployment would silently inherit the backend model's real cache rates.
|
|
Every backend here publishes non-zero ones, which is what makes the assertion mean
|
|
something."""
|
|
cache_fields = (
|
|
"cache_creation_input_token_cost",
|
|
"cache_creation_input_token_cost_above_1hr",
|
|
"cache_creation_input_token_cost_above_200k_tokens",
|
|
"cache_read_input_token_cost",
|
|
"cache_read_input_token_cost_above_200k_tokens",
|
|
)
|
|
builtin = litellm.get_model_info(model=backend)
|
|
assert any(builtin.get(field) for field in cache_fields), "backend publishes no cache pricing to leak"
|
|
|
|
router = _ptu_router(litellm_params={"model": backend})
|
|
priced = litellm.model_cost[router.model_list[0]["model_info"]["id"]]
|
|
|
|
assert [field for field in cache_fields if priced.get(field)] == []
|
|
|
|
|
|
def test_zeroing_a_ptu_deployment_leaves_its_backend_model_priced():
|
|
"""A sibling deployment on the same backend must keep billing normally."""
|
|
backend = "anthropic/claude-sonnet-4-5-20250929"
|
|
builtin = litellm.get_model_info(model=backend)["input_cost_per_token"]
|
|
assert builtin > 0
|
|
|
|
_ptu_router(litellm_params={"model": backend})
|
|
|
|
assert litellm.get_model_info(model=backend)["input_cost_per_token"] == builtin
|
|
|
|
|
|
def test_the_registered_id_is_the_one_the_operator_declared():
|
|
"""Registration must key the deployment by the declared id, not by a hash of params that
|
|
zeroing has just rewritten. The id keys cooldowns, budgets and every spend row already
|
|
written, so minting one here would move all of them.
|
|
|
|
A derived id is no longer reachable for a reservation: zeroing requires PTU terms and
|
|
PTU terms now require a declared id, so the two never combine."""
|
|
params = {"input_cost_per_token": 5e-06}
|
|
priced = _ptu_router(litellm_params=params, ptu_enabled=False).model_list[0]["model_info"]["id"]
|
|
zeroed = _ptu_router(litellm_params=params).model_list[0]["model_info"]["id"]
|
|
|
|
assert priced == zeroed == "ptu-alpha-eastus"
|
|
|
|
|
|
def test_a_database_backed_deployment_is_left_alone():
|
|
"""The write endpoints already zero those, and they answer 400 rather than silently
|
|
rewriting a rate the caller sent."""
|
|
entry = _ptu_router(model_info={**_PTU_MODEL_INFO, "db_model": True}).model_list[0]
|
|
|
|
assert entry["litellm_params"].get("input_cost_per_token") is None
|
|
|
|
|
|
def test_nothing_is_zeroed_while_the_feature_is_off():
|
|
"""No flat cost accrues with the flag off, so zeroing would serve the traffic free."""
|
|
entry = _ptu_router(litellm_params={"input_cost_per_token": 5e-06}, ptu_enabled=False).model_list[0]
|
|
|
|
assert entry["litellm_params"]["input_cost_per_token"] == 5e-06
|
|
|
|
|
|
@pytest.mark.parametrize("dropped", ["team_id", "ptu_effective_from"], ids=["no team_id", "no ptu_effective_from"])
|
|
def test_an_incomplete_reservation_is_refused_rather_than_served(dropped):
|
|
"""POST /model/new answers 400 for exactly this config, so config.yaml must not quietly
|
|
accept it. Serving it would bill per token while accruing no flat cost, which is the
|
|
state the operator was trying to leave."""
|
|
incomplete = {k: v for k, v in _PTU_MODEL_INFO.items() if k != dropped}
|
|
|
|
with pytest.raises(ValueError, match="PTU configuration on model 'gpt") as raised:
|
|
_ptu_router(model_info=incomplete, litellm_params={"input_cost_per_token": 5e-06})
|
|
|
|
assert "gpt-4o-ptu" in str(raised.value)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"dropped, expected",
|
|
[
|
|
("team_id", "team_id is required when PTU fields are set (one model maps to one team)"),
|
|
("cost_per_ptu_per_hour", "ptu_count and cost_per_ptu_per_hour must be set together"),
|
|
],
|
|
ids=["no team_id", "count without rate"],
|
|
)
|
|
def test_the_refusal_reason_is_the_one_the_model_endpoint_answers_with(dropped, expected):
|
|
"""One rule, stated once. If these drift, an operator gets contradictory guidance
|
|
depending on which path they used."""
|
|
incomplete = {k: v for k, v in _PTU_MODEL_INFO.items() if k != dropped}
|
|
|
|
assert ptu_config_error(incomplete) == expected
|
|
with pytest.raises(ValueError, match="PTU configuration on model 'gpt") as raised:
|
|
_ptu_router(model_info=incomplete)
|
|
|
|
assert expected in str(raised.value)
|
|
|
|
|
|
@pytest.mark.parametrize("dropped", ["team_id", "ptu_effective_from"], ids=["no team_id", "no ptu_effective_from"])
|
|
def test_an_incomplete_reservation_is_left_alone_while_the_feature_is_off(dropped):
|
|
"""Nothing accrues with the flag off, so refusing a deployment there would take a
|
|
serving model away from an operator who never opted in."""
|
|
incomplete = {k: v for k, v in _PTU_MODEL_INFO.items() if k != dropped}
|
|
entry = _ptu_router(
|
|
model_info=incomplete, litellm_params={"input_cost_per_token": 5e-06}, ptu_enabled=False
|
|
).model_list[0]
|
|
|
|
assert entry["litellm_params"]["input_cost_per_token"] == 5e-06
|
|
|
|
|
|
@pytest.mark.parametrize("dropped", ["team_id", "ptu_effective_from"], ids=["no team_id", "no ptu_effective_from"])
|
|
def test_the_proxy_drops_the_deployment_rather_than_failing_to_boot(dropped):
|
|
"""The proxy builds its router with ignore_invalid_deployments, so one bad entry must
|
|
cost that entry and not the whole config."""
|
|
incomplete = {k: v for k, v in _PTU_MODEL_INFO.items() if k != dropped}
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": "True"}, clear=False):
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "gpt-4o-ptu",
|
|
"litellm_params": {"model": "anthropic/claude-sonnet-4-5-20250929", "api_key": "sk-not-used"},
|
|
"model_info": dict(incomplete),
|
|
},
|
|
{
|
|
"model_name": "plain-sibling",
|
|
"litellm_params": {"model": "anthropic/claude-sonnet-4-5-20250929", "api_key": "sk-not-used"},
|
|
},
|
|
],
|
|
ignore_invalid_deployments=True,
|
|
)
|
|
|
|
assert [entry["model_name"] for entry in router.model_list] == ["plain-sibling"]
|
|
|
|
|
|
def test_a_complete_reservation_still_registers():
|
|
"""The refusal must be scoped to a broken reservation, not to PTU configuration."""
|
|
entry = _ptu_router().model_list[0]
|
|
|
|
assert entry["model_name"] == "gpt-4o-ptu"
|
|
assert entry["litellm_params"]["input_cost_per_token"] == 0.0
|
|
|
|
|
|
def test_nested_custom_model_info_does_not_pollute_shared_backend():
|
|
backend_model = "gpt-4o-search-preview"
|
|
custom_id = "lit5471-search-custom"
|
|
sibling_id = "lit5471-search-sibling"
|
|
builtin_info = copy.deepcopy(litellm.get_model_info(model=backend_model))
|
|
expected_nested = copy.deepcopy(builtin_info["search_context_cost_per_query"])
|
|
model_keys = {
|
|
backend_model: copy.deepcopy(litellm.model_cost.get(backend_model)),
|
|
custom_id: copy.deepcopy(litellm.model_cost.get(custom_id)),
|
|
sibling_id: copy.deepcopy(litellm.model_cost.get(sibling_id)),
|
|
}
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "search-custom",
|
|
"litellm_params": {"model": backend_model, "api_key": "fake-key"},
|
|
"model_info": {
|
|
"id": custom_id,
|
|
"search_context_cost_per_query": {
|
|
"search_context_size_low": 0.123,
|
|
},
|
|
},
|
|
},
|
|
{
|
|
"model_name": "search-sibling",
|
|
"litellm_params": {"model": backend_model, "api_key": "fake-key"},
|
|
"model_info": {"id": sibling_id},
|
|
},
|
|
],
|
|
)
|
|
|
|
custom_info = router.get_deployment_model_info(model_id=custom_id, model_name=backend_model)
|
|
sibling_info = router.get_deployment_model_info(model_id=sibling_id, model_name=backend_model)
|
|
|
|
assert custom_info is not None
|
|
assert custom_info["search_context_cost_per_query"]["search_context_size_low"] == 0.123
|
|
assert litellm.model_cost[backend_model]["search_context_cost_per_query"] == expected_nested
|
|
assert sibling_info is not None
|
|
assert sibling_info["search_context_cost_per_query"] == expected_nested
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_base_model_custom_info_does_not_pollute_cached_base_model():
|
|
base_model = "azure/gpt-4o"
|
|
deployment_id = "lit5471-base-model"
|
|
base_model_info = copy.deepcopy(litellm.get_model_info(model=base_model))
|
|
model_keys = {
|
|
"azure/gpt-4o": copy.deepcopy(litellm.model_cost.get("azure/gpt-4o")),
|
|
deployment_id: copy.deepcopy(litellm.model_cost.get(deployment_id)),
|
|
}
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "azure-custom",
|
|
"litellm_params": {
|
|
"model": "gpt-4o",
|
|
"custom_llm_provider": "azure",
|
|
"api_key": "fake-key",
|
|
},
|
|
"model_info": {
|
|
"id": deployment_id,
|
|
"base_model": base_model,
|
|
"input_cost_per_token": 0.777,
|
|
},
|
|
}
|
|
],
|
|
)
|
|
|
|
info = router.get_deployment_model_info(model_id=deployment_id, model_name=base_model)
|
|
|
|
assert info is not None
|
|
assert info["input_cost_per_token"] == 0.777
|
|
assert litellm.get_model_info(model=base_model) == base_model_info
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_builtin_only_deployment_info_is_not_the_cached_object():
|
|
backend_model = "gpt-4o-search-preview"
|
|
deployment_id = "lit5471-builtin-only"
|
|
litellm.get_model_info.cache_clear()
|
|
model_keys = {deployment_id: copy.deepcopy(litellm.model_cost.get(deployment_id))}
|
|
try:
|
|
cached_info = litellm.get_model_info(model=backend_model)
|
|
assert cached_info["search_context_cost_per_query"]
|
|
|
|
info = Router(model_list=[]).get_deployment_model_info(model_id=deployment_id, model_name=backend_model)
|
|
|
|
assert info is not None
|
|
assert info["search_context_cost_per_query"] == cached_info["search_context_cost_per_query"]
|
|
assert _nested_container_ids(info).isdisjoint(_nested_container_ids(cached_info))
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_custom_only_deployment_info_is_not_the_registry_entry():
|
|
unknown_backend = "openai/lit5471-unknown-backend"
|
|
deployment_id = "lit5471-custom-only"
|
|
nested_pricing = {"search_context_size_low": 0.123}
|
|
model_keys = {
|
|
unknown_backend: copy.deepcopy(litellm.model_cost.get(unknown_backend)),
|
|
deployment_id: copy.deepcopy(litellm.model_cost.get(deployment_id)),
|
|
}
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "custom-only",
|
|
"litellm_params": {"model": unknown_backend, "api_key": "fake-key"},
|
|
"model_info": {"id": deployment_id, "search_context_cost_per_query": dict(nested_pricing)},
|
|
}
|
|
],
|
|
)
|
|
registry_entry = litellm.model_cost[deployment_id]
|
|
|
|
info = router.get_deployment_model_info(model_id=deployment_id, model_name=unknown_backend)
|
|
|
|
assert info is not None
|
|
assert info["search_context_cost_per_query"] == nested_pricing
|
|
assert _nested_container_ids(info).isdisjoint(_nested_container_ids(registry_entry))
|
|
finally:
|
|
_restore_model_cost_entries(model_keys)
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
def test_router_model_info_deep_copies_nested_cached_metadata():
|
|
model = "openai/gpt-4o-search-preview"
|
|
litellm.get_model_info.cache_clear()
|
|
try:
|
|
cached_info = litellm.get_model_info(model=model)
|
|
assert cached_info is not None
|
|
expected_nested = copy.deepcopy(cached_info["search_context_cost_per_query"])
|
|
assert expected_nested
|
|
|
|
router = Router(model_list=[])
|
|
merged_info = router.get_router_model_info(
|
|
deployment={
|
|
"model_name": "search",
|
|
"litellm_params": {"model": "gpt-4o-search-preview"},
|
|
"model_info": {"id": "lit5471-router-model-info"},
|
|
},
|
|
received_model_name="search",
|
|
)
|
|
|
|
assert merged_info["search_context_cost_per_query"] == expected_nested
|
|
assert _nested_container_ids(merged_info).isdisjoint(_nested_container_ids(cached_info))
|
|
assert litellm.get_model_info(model=model)["search_context_cost_per_query"] == expected_nested
|
|
finally:
|
|
litellm.get_model_info.cache_clear()
|
|
|
|
|
|
# --- a config.yaml reservation must carry an id its operator owns --------------------
|
|
|
|
|
|
def test_a_reservation_without_a_declared_id_is_refused():
|
|
"""Left underived the id is a hash of the resolved litellm_params, so rotating the
|
|
credential mints a second identity and the catch-up bills the window again under it.
|
|
The flat cost is keyed by that id and a written charge is never retracted, so the
|
|
duplicate is permanent."""
|
|
anonymous = {k: v for k, v in _PTU_MODEL_INFO.items() if k != "id"}
|
|
|
|
with pytest.raises(ValueError, match=re.escape("model_info.id is required")):
|
|
_ptu_router(model_info=anonymous)
|
|
|
|
|
|
def test_the_id_rule_does_not_reach_a_deployment_without_ptu_config():
|
|
"""An ordinary deployment keeps deriving its id, which is most of every config.yaml."""
|
|
entry = _ptu_router(model_info={"team_id": "team-alpha"}).model_list[0]
|
|
|
|
assert entry["model_info"]["id"]
|
|
|
|
|
|
def test_a_reservation_is_left_alone_while_the_feature_is_off():
|
|
anonymous = {k: v for k, v in _PTU_MODEL_INFO.items() if k != "id"}
|
|
entry = _ptu_router(model_info=anonymous, ptu_enabled=False).model_list[0]
|
|
|
|
assert entry["model_info"]["id"]
|
|
|
|
|
|
def test_two_reservations_cannot_share_one_id():
|
|
"""Both would key the same sentinel row, so the second upsert overwrites the first and
|
|
one reservation is billed at the other's rate."""
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": "True"}, clear=False):
|
|
with pytest.raises(ValueError, match="declared on more than one deployment"):
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "azure-ptu",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://e.azure.com"},
|
|
"model_info": dict(_PTU_MODEL_INFO),
|
|
},
|
|
{
|
|
"model_name": "azure-ptu-west",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://w.azure.com"},
|
|
"model_info": dict(_PTU_MODEL_INFO),
|
|
},
|
|
]
|
|
)
|
|
|
|
|
|
def test_two_reservations_with_distinct_ids_both_register():
|
|
"""The refusal must be scoped to a collision, not to a team running two regions."""
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": "True"}, clear=False):
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "azure-ptu",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://e.azure.com"},
|
|
"model_info": dict(_PTU_MODEL_INFO),
|
|
},
|
|
{
|
|
"model_name": "azure-ptu-west",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://w.azure.com"},
|
|
"model_info": {**_PTU_MODEL_INFO, "id": "ptu-alpha-westus"},
|
|
},
|
|
]
|
|
)
|
|
|
|
assert sorted(m["model_info"]["id"] for m in router.model_list) == ["ptu-alpha-eastus", "ptu-alpha-westus"]
|
|
|
|
|
|
@pytest.mark.parametrize("declared", ["dup-id", 12345], ids=["string id", "numeric id"])
|
|
def test_a_duplicate_id_is_caught_whatever_yaml_parsed_it_as(declared):
|
|
"""An unquoted id in config.yaml arrives as an int, and ModelInfo stores it as a string,
|
|
so both deployments would still key one flat-cost row."""
|
|
|
|
def entry(name, region):
|
|
return {
|
|
"model_name": name,
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": f"https://{region}.azure.com"},
|
|
"model_info": {**_PTU_MODEL_INFO, "id": declared},
|
|
}
|
|
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": "True"}, clear=False):
|
|
with pytest.raises(ValueError, match="declared on more than one deployment"):
|
|
Router(model_list=[entry("a", "eastus"), entry("b", "westus")])
|
|
|
|
|
|
def test_a_bare_yaml_date_bound_does_not_escape_the_id_rule():
|
|
"""`ptu_effective_to: 2027-01-01` unquoted loads as a date. While that failed to parse,
|
|
the reservation was invisible to PTU entirely: no id rule, no zeroing, no flat cost."""
|
|
import datetime as _dt
|
|
|
|
windowed = {k: v for k, v in _PTU_MODEL_INFO.items() if k != "id"}
|
|
|
|
with pytest.raises(ValueError, match=re.escape("model_info.id is required")):
|
|
_ptu_router(model_info={**windowed, "ptu_effective_to": _dt.date(2027, 1, 1)})
|
|
|
|
|
|
def test_a_reservation_declaring_id_zero_registers():
|
|
"""0 is stable and unique, so reading it as absent refused a correct config."""
|
|
entry = _ptu_router(model_info={**_PTU_MODEL_INFO, "id": 0}).model_list[0]
|
|
|
|
assert entry["model_info"]["id"] == "0"
|
|
|
|
|
|
def test_a_falsy_id_is_still_scanned_for_collisions():
|
|
"""The duplicate scan skipped falsy ids, so a reservation on '0' could share its key with
|
|
an ordinary deployment and the id index would keep only the last one registered."""
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": "True"}, clear=False):
|
|
with pytest.raises(ValueError, match="declared on more than one deployment"):
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "azure-ptu",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://e.azure.com"},
|
|
"model_info": {**_PTU_MODEL_INFO, "id": "0"},
|
|
},
|
|
{
|
|
"model_name": "plain-sibling",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://w.azure.com"},
|
|
"model_info": {"id": 0},
|
|
},
|
|
]
|
|
)
|
|
|
|
|
|
# --- a reservation declared while the feature is off says so ------------------------
|
|
|
|
|
|
def _ptu_warnings(caplog):
|
|
return tuple(
|
|
record.getMessage()
|
|
for record in caplog.records
|
|
if record.name == "LiteLLM Router" and record.levelno == logging.WARNING and "PTU" in record.getMessage()
|
|
)
|
|
|
|
|
|
def test_a_reservation_declared_while_the_feature_is_off_is_warned_about(caplog):
|
|
"""The deployment serves and bills per token, so without this the operator believes they
|
|
reserved capacity and sees no signal anywhere that nothing accrues."""
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM Router"):
|
|
_ptu_router(ptu_enabled=False)
|
|
|
|
warnings = _ptu_warnings(caplog)
|
|
|
|
assert len(warnings) == 1
|
|
assert "gpt-4o-ptu" in warnings[0]
|
|
assert "LITELLM_ENABLE_PTU_COST_ATTRIBUTION" in warnings[0]
|
|
|
|
|
|
def test_a_reservation_is_not_warned_about_while_the_feature_is_on(caplog):
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM Router"):
|
|
_ptu_router()
|
|
|
|
assert _ptu_warnings(caplog) == ()
|
|
|
|
|
|
def test_a_deployment_carrying_no_ptu_field_is_not_warned_about(caplog):
|
|
"""Most of every config.yaml, so warning here would fire on proxies that never asked."""
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM Router"):
|
|
_ptu_router(model_info={"team_id": "team-alpha"}, ptu_enabled=False)
|
|
|
|
assert _ptu_warnings(caplog) == ()
|
|
|
|
|
|
def test_a_half_written_reservation_is_warned_about(caplog):
|
|
"""A count with no rate is not a chargeable reservation, but the operator still meant to
|
|
declare one, so what they wrote is what decides whether they hear about it."""
|
|
half_written = {k: v for k, v in _PTU_MODEL_INFO.items() if k != "cost_per_ptu_per_hour"}
|
|
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM Router"):
|
|
_ptu_router(model_info=half_written, ptu_enabled=False)
|
|
|
|
assert len(_ptu_warnings(caplog)) == 1
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"typo",
|
|
[
|
|
{"ptu_count": 0},
|
|
{"ptu_count": 0, "cost_per_ptu_per_hour": 0, "ptu_effective_from": None},
|
|
],
|
|
ids=["count out of range", "every value still a zero placeholder"],
|
|
)
|
|
def test_a_reservation_dropped_by_a_typo_is_warned_about(caplog, typo):
|
|
"""An out-of-range value fails ModelInfo before the flag is ever consulted, so the
|
|
deployment stops serving on a proxy that never enabled PTU. The warning is what tells the
|
|
operator which feature the entry that vanished belonged to.
|
|
|
|
Built the way proxy_server builds it, since dropping rather than raising is what
|
|
``ignore_invalid_deployments`` does and config.yaml is loaded with it on.
|
|
"""
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": ""}, clear=False):
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM Router"):
|
|
router = Router(
|
|
ignore_invalid_deployments=True,
|
|
model_list=[
|
|
{
|
|
"model_name": "gpt-4o-ptu",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://e.azure.com"},
|
|
"model_info": {**_PTU_MODEL_INFO, **typo},
|
|
}
|
|
],
|
|
)
|
|
|
|
assert router.model_list == []
|
|
assert len(_ptu_warnings(caplog)) == 1
|
|
|
|
|
|
def test_a_db_backed_reservation_is_not_warned_about(caplog):
|
|
"""/model/new already answered the caller with a 400, so repeating it on every reload
|
|
would report the operator's own rejected write back to them as a standing problem."""
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM Router"):
|
|
_ptu_router(model_info={**_PTU_MODEL_INFO, "db_model": True}, ptu_enabled=False)
|
|
|
|
assert _ptu_warnings(caplog) == ()
|
|
|
|
|
|
def test_every_declaring_deployment_is_named(caplog):
|
|
"""One line naming all of them, so a reload does not bury the config in repeats."""
|
|
with patch.dict(os.environ, {"LITELLM_ENABLE_PTU_COST_ATTRIBUTION": ""}, clear=False):
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM Router"):
|
|
Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "azure-ptu-east",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://e.azure.com"},
|
|
"model_info": dict(_PTU_MODEL_INFO),
|
|
},
|
|
{
|
|
"model_name": "azure-ptu-west",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://w.azure.com"},
|
|
"model_info": {**_PTU_MODEL_INFO, "id": "ptu-alpha-westus"},
|
|
},
|
|
{
|
|
"model_name": "plain-gpt-4o",
|
|
"litellm_params": {"model": "azure/gpt-4o", "api_key": "k", "api_base": "https://p.azure.com"},
|
|
"model_info": {"id": "plain"},
|
|
},
|
|
]
|
|
)
|
|
|
|
warnings = _ptu_warnings(caplog)
|
|
|
|
assert len(warnings) == 1
|
|
assert "azure-ptu-east" in warnings[0]
|
|
assert "azure-ptu-west" in warnings[0]
|
|
assert "plain-gpt-4o" not in warnings[0]
|
|
|
|
|
|
def _simulate_price_data_reload_with_provider_sets(monkeypatch, fetched_catalog):
|
|
"""Like `_simulate_price_data_reload`, plus the provider model-set refresh the proxy's
|
|
`_swap_in_model_cost_map` does before replaying, so bare names in the new catalog resolve."""
|
|
monkeypatch.setattr(litellm, "model_cost", fetched_catalog)
|
|
_invalidate_model_cost_lowercase_map()
|
|
litellm.add_known_models(model_cost_map=fetched_catalog)
|
|
reapply_runtime_model_cost_registrations()
|
|
|
|
|
|
def test_a_config_deployment_dropped_by_a_stale_cost_map_comes_back_on_reload(monkeypatch):
|
|
"""
|
|
Booting on the bundled backup, a bare model that only the remote catalog knows
|
|
cannot be provider-resolved, so the proxy router (ignore_invalid_deployments) drops
|
|
it. Once a reload brings in a catalog that knows the model, the deployment must be
|
|
served again with its access groups, and exactly once however many reloads follow.
|
|
"""
|
|
backend = "lit-5766-only-in-remote-catalog"
|
|
try:
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "new-model",
|
|
"litellm_params": {"model": backend, "api_key": "k"},
|
|
"model_info": {"id": "new-id", "access_groups": ["team-models"]},
|
|
},
|
|
{
|
|
"model_name": "control-model",
|
|
"litellm_params": {"model": "hosted_vllm/control-backend", "api_key": "k"},
|
|
"model_info": {"id": "control-id", "access_groups": ["team-models"]},
|
|
},
|
|
],
|
|
ignore_invalid_deployments=True,
|
|
)
|
|
assert router.get_model_names() == ["control-model"]
|
|
assert router.get_model_access_groups(model_name="new-model") == {}
|
|
|
|
fresh_catalog = {**litellm.model_cost, backend: {"litellm_provider": "openai", "mode": "chat"}}
|
|
_simulate_price_data_reload_with_provider_sets(monkeypatch, fresh_catalog)
|
|
_simulate_price_data_reload_with_provider_sets(monkeypatch, fresh_catalog)
|
|
|
|
assert sorted(router.get_model_names()) == ["control-model", "new-model"]
|
|
assert router.get_model_access_groups(model_name="new-model") == {"team-models": ["new-model"]}
|
|
assert [d["model_info"]["id"] for d in router.model_list] == ["control-id", "new-id"]
|
|
assert "new-id" in litellm.model_cost
|
|
finally:
|
|
litellm.open_ai_chat_completion_models.discard(backend)
|
|
litellm.models_by_provider["openai"].discard(backend)
|
|
|
|
|
|
def test_a_config_deployment_dropped_for_a_permanent_reason_is_not_retried_on_reload(monkeypatch):
|
|
"""
|
|
Only provider-resolution drops can be healed by a fresh catalog. A deployment that
|
|
fails after its provider resolved (here a pass-through vertex entry with no project)
|
|
has already touched router state, so replaying it on every reload would leak into
|
|
`deployment_names` each time.
|
|
"""
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "vertex-passthrough",
|
|
"litellm_params": {"model": "vertex_ai/gemini-2.5-flash", "use_in_pass_through": True},
|
|
"model_info": {"id": "vertex-id"},
|
|
},
|
|
{
|
|
"model_name": "control-model",
|
|
"litellm_params": {"model": "hosted_vllm/control-backend", "api_key": "k"},
|
|
"model_info": {"id": "control-id"},
|
|
},
|
|
],
|
|
ignore_invalid_deployments=True,
|
|
)
|
|
assert router.get_model_names() == ["control-model"]
|
|
names_after_boot = list(router.deployment_names)
|
|
|
|
_simulate_price_data_reload_with_provider_sets(monkeypatch, dict(litellm.model_cost))
|
|
|
|
assert router.get_model_names() == ["control-model"]
|
|
assert router.deployment_names == names_after_boot
|
|
|
|
|
|
def test_price_data_reload_refreshes_the_cached_model_group_and_deployment_info(monkeypatch):
|
|
"""
|
|
Budget reservation reads pricing through the router's lru-cached group and
|
|
deployment lookups. A reload swaps the catalog without touching model_list, so
|
|
unless the replay clears those caches the next reservation prices against the
|
|
old catalog until some unrelated model-list change happens to evict it.
|
|
"""
|
|
router = Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "grp",
|
|
"litellm_params": {"model": "openai/gpt-4o", "api_key": "k"},
|
|
"model_info": {"id": "dep-a"},
|
|
}
|
|
]
|
|
)
|
|
old_price = router.cached_model_group_info("grp").input_cost_per_token
|
|
assert router.cached_deployment_model_info("dep-a", "openai/gpt-4o")["input_cost_per_token"] == old_price
|
|
|
|
new_price = old_price * 10
|
|
fresh_catalog = copy.deepcopy(litellm.model_cost)
|
|
fresh_catalog["gpt-4o"]["input_cost_per_token"] = new_price
|
|
_simulate_price_data_reload_with_provider_sets(monkeypatch, fresh_catalog)
|
|
|
|
assert router.cached_model_group_info("grp").input_cost_per_token == new_price
|
|
assert router.cached_deployment_model_info("dep-a", "openai/gpt-4o")["input_cost_per_token"] == new_price
|