From 430fb71933f559b948a0082adfaf027e0c3f42e5 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Wed, 26 Aug 2026 00:20:35 -0700 Subject: [PATCH 1/2] test(cost-estimate): pin the prices and period totals /cost/estimate returns The endpoint already had tests for deployments that set both an input and an output price, and for litellm_params winning over model_info. Nothing covered a deployment that prices only one of the two sides, the daily and monthly totals, or the price and provider read from the public cost map. Found by changing one line of cost_tracking_settings.py at a time and running the mapped test file against each change. Nine of eleven one-line changes went unnoticed: dropping custom pricing entirely when only one side is priced, billing the unpriced side at something other than zero, skipping the model lookup so the reported price and provider go empty, turning zero requests a day into a cost of zero rather than no estimate, and scaling a period total by one request instead of the real count. The seven tests added here kill all eleven. The cost math is real; only the router is faked, matching the fixtures already in this file. --- .../test_cost_tracking_settings.py | 130 ++++++++++++++++++ 1 file changed, 130 insertions(+) diff --git a/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py b/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py index e1eb031abc2..2b2476d3c99 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py +++ b/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py @@ -780,3 +780,133 @@ class TestBlockRequestsForModelsWithoutPricing: assert response.status_code == 500 assert "error" in response.json()["detail"] + + +AN_ALIAS = "onprem/alias" +AN_UNDERLYING_MODEL = "vendor/model" +A_MAPPED_MODEL = "openai/mapped-only-model" +INPUT_TOKENS = 1000 +OUTPUT_TOKENS = 500 + + +def _router_pricing(**pricing: float) -> MagicMock: + mock_router = MagicMock() + mock_router.get_model_list.return_value = [ + { + "model_name": AN_ALIAS, + "litellm_params": { + "model": AN_UNDERLYING_MODEL, + "custom_llm_provider": "openai", + **pricing, + }, + "model_info": {}, + } + ] + return mock_router + + +async def _estimate(mock_router: MagicMock | None, model: str = AN_ALIAS, **overrides: int): + from litellm.proxy._types import CostEstimateRequest + from litellm.proxy.management_endpoints.cost_tracking_settings import estimate_cost + + request = CostEstimateRequest( + model=model, + input_tokens=INPUT_TOKENS, + output_tokens=OUTPUT_TOKENS, + **overrides, + ) + with patch("litellm.proxy.proxy_server.llm_router", mock_router): + return await estimate_cost(request=request, user_api_key_dict=MagicMock()) + + +class TestEstimateCostPartiallyPricedDeployments: + @pytest.mark.asyncio + async def test_a_deployment_that_prices_only_input_bills_output_at_zero(self): + response = await _estimate(_router_pricing(input_cost_per_token=0.000001)) + + assert response.input_cost_per_token == pytest.approx(0.000001) + assert response.output_cost_per_token == 0.0 + assert response.cost_per_request == pytest.approx(0.001) + + @pytest.mark.asyncio + async def test_a_deployment_that_prices_only_output_bills_input_at_zero(self): + response = await _estimate(_router_pricing(output_cost_per_token=0.000002)) + + assert response.input_cost_per_token == 0.0 + assert response.output_cost_per_token == pytest.approx(0.000002) + assert response.cost_per_request == pytest.approx(0.001) + + @pytest.mark.asyncio + async def test_a_model_priced_only_by_the_cost_map_reports_that_price_and_provider(self): + saved_model_cost = dict(litellm.model_cost) + litellm.register_model( + { + A_MAPPED_MODEL: { + "input_cost_per_token": 0.000005, + "output_cost_per_token": 0.000006, + "litellm_provider": "openai", + "mode": "chat", + } + } + ) + try: + response = await _estimate(None, model=A_MAPPED_MODEL) + finally: + litellm.model_cost = saved_model_cost + + assert response.input_cost_per_token == pytest.approx(0.000005) + assert response.output_cost_per_token == pytest.approx(0.000006) + assert response.provider == "openai" + + +class TestEstimateCostPeriodTotals: + @pytest.mark.asyncio + async def test_zero_requests_a_day_reports_no_daily_cost_rather_than_zero(self): + response = await _estimate( + _router_pricing(input_cost_per_token=0.000001, output_cost_per_token=0.000002), + num_requests_per_day=0, + ) + + assert response.daily_cost is None + assert response.daily_input_cost is None + assert response.daily_output_cost is None + + @pytest.mark.asyncio + async def test_daily_totals_scale_every_component_by_the_request_count(self): + response = await _estimate( + _router_pricing(input_cost_per_token=0.000001, output_cost_per_token=0.000002), + num_requests_per_day=100, + ) + + assert response.input_cost_per_request == pytest.approx(0.001) + assert response.output_cost_per_request == pytest.approx(0.001) + assert response.daily_input_cost == pytest.approx(0.1) + assert response.daily_output_cost == pytest.approx(0.1) + assert response.daily_cost == pytest.approx(0.2) + + @pytest.mark.asyncio + async def test_a_month_and_a_day_are_totalled_from_their_own_request_counts(self): + response = await _estimate( + _router_pricing(input_cost_per_token=0.000001, output_cost_per_token=0.000002), + num_requests_per_day=100, + num_requests_per_month=3000, + ) + + assert response.daily_cost == pytest.approx(0.2) + assert response.monthly_cost == pytest.approx(6.0) + assert response.monthly_input_cost == pytest.approx(3.0) + assert response.monthly_output_cost == pytest.approx(3.0) + + @pytest.mark.asyncio + async def test_a_configured_margin_is_totalled_per_period_like_the_other_components(self, monkeypatch): + monkeypatch.setattr(litellm, "cost_margin_config", {"openai": 0.10}) + + response = await _estimate( + _router_pricing(input_cost_per_token=0.000001, output_cost_per_token=0.000002), + num_requests_per_day=100, + ) + + assert response.margin_cost_per_request == pytest.approx(0.0002) + assert response.cost_per_request == pytest.approx(0.0022) + assert response.daily_margin_cost == pytest.approx(0.02) + assert response.daily_cost == pytest.approx(0.22) From 0c87bf5de9aa87d176c2fec3af078455261e5f6d Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Wed, 26 Aug 2026 00:32:25 -0700 Subject: [PATCH 2/2] test(cost-estimate): keep the new tests inside the test-quality budgets Two of the new lines tripped the ratcheting gate. TQ005 flagged restoring litellm.model_cost by assignment. Dropped the save/restore pair for monkeypatch.setitem, which adds the one model the test needs and takes it back out at teardown, so the module global is never reassigned. TQ008 flagged patching litellm.proxy.proxy_server.llm_router. The endpoint imports the router from that module inside the function body, so there is no seam to inject through without changing the endpoint. Suppressed with the reason already used elsewhere in the suite for the same module global, on the single helper the new tests share. --- .../test_cost_tracking_settings.py | 29 +++++++++---------- 1 file changed, 14 insertions(+), 15 deletions(-) diff --git a/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py b/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py index 2b2476d3c99..ec62cc47018 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py +++ b/tests/test_litellm/proxy/management_endpoints/test_cost_tracking_settings.py @@ -815,7 +815,9 @@ async def _estimate(mock_router: MagicMock | None, model: str = AN_ALIAS, **over output_tokens=OUTPUT_TOKENS, **overrides, ) - with patch("litellm.proxy.proxy_server.llm_router", mock_router): + with patch( # test-quality-ok: proxy_server module global is the endpoint's only injection point + "litellm.proxy.proxy_server.llm_router", mock_router + ): return await estimate_cost(request=request, user_api_key_dict=MagicMock()) @@ -837,22 +839,19 @@ class TestEstimateCostPartiallyPricedDeployments: assert response.cost_per_request == pytest.approx(0.001) @pytest.mark.asyncio - async def test_a_model_priced_only_by_the_cost_map_reports_that_price_and_provider(self): - saved_model_cost = dict(litellm.model_cost) - litellm.register_model( + async def test_a_model_priced_only_by_the_cost_map_reports_that_price_and_provider(self, monkeypatch): + monkeypatch.setitem( + litellm.model_cost, + A_MAPPED_MODEL, { - A_MAPPED_MODEL: { - "input_cost_per_token": 0.000005, - "output_cost_per_token": 0.000006, - "litellm_provider": "openai", - "mode": "chat", - } - } + "input_cost_per_token": 0.000005, + "output_cost_per_token": 0.000006, + "litellm_provider": "openai", + "mode": "chat", + }, ) - try: - response = await _estimate(None, model=A_MAPPED_MODEL) - finally: - litellm.model_cost = saved_model_cost + + response = await _estimate(None, model=A_MAPPED_MODEL) assert response.input_cost_per_token == pytest.approx(0.000005) assert response.output_cost_per_token == pytest.approx(0.000006)