From 727905dfc31b67484cc8f3650ff3495aa02936e5 Mon Sep 17 00:00:00 2001 From: Marty Sullivan Date: Sun, 16 Aug 2026 16:43:56 -0400 Subject: [PATCH] fix(batches): only use deployment pricing when the deployment declares it The router registers a model_info entry for every deployment, priced or not, and get_model_info fills absent costs with 0. Resolving deployment pricing through it therefore reported a free deployment for any ordinary one, which priced its batches at $0 while usage stayed correct: the same silent under-count this branch set out to remove, widened from bedrock to every provider. Caught by a live batch run, where four vertex batches that price correctly today came back at $0. The raw registration is now what decides: pricing is used only when the deployment actually declares one of the batch cost fields, so ordinary deployments fall back to the global cost map exactly as before. The earlier test missed this by using a deployment id that was never registered, where get_model_info does raise; a real deployment is always registered. --- litellm/litellm_core_utils/litellm_logging.py | 16 ++++++++++++++-- .../litellm_core_utils/test_litellm_logging.py | 18 ++++++++++++++++++ 2 files changed, 32 insertions(+), 2 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 662ce38f746..f6f3885bff2 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -584,14 +584,26 @@ class Logging(LiteLLMLoggingBaseClass): """Pricing the router registered under this deployment's model_info.id. Returns None when the deployment declares no pricing of its own, so the - caller falls back to the global cost map. + caller falls back to the global cost map. The raw registration is what + decides that: the router registers an entry for every deployment, and + get_model_info fills absent costs with 0, so asking it directly cannot + tell "configured as free" apart from "no pricing configured". """ + pricing_keys: Final = ( + "input_cost_per_token", + "output_cost_per_token", + "input_cost_per_token_batches", + "output_cost_per_token_batches", + ) model_id: Final = self.get_router_model_id() if model_id is None: return None + registered: Final = litellm.model_cost.get(model_id) + if not isinstance(registered, dict) or not any(registered.get(key) is not None for key in pricing_keys): + return None try: return litellm.get_model_info(model=model_id) - except Exception: # noqa: BLE001 # get_model_info raises for any id with no registered pricing + except Exception: # noqa: BLE001 # get_model_info raises for ids it cannot resolve a provider for return None def update_environment_variables( diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index e9dc65e526d..3b5e8aaf1f8 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -367,6 +367,24 @@ class TestGetRouterDeploymentModelInfo: logging_obj.litellm_params = {"litellm_metadata": {"model_info": {"id": "deploy-never-registered"}}} assert logging_obj.get_router_deployment_model_info() is None + def test_returns_none_when_deployment_registered_without_pricing(self, logging_obj): + """The router registers an entry for EVERY deployment, priced or not. + + get_model_info fills absent costs with 0, so consulting it directly would + hand back free pricing for an ordinary deployment and bill its batches $0. + """ + deployment_id = "deploy-no-pricing-1" + litellm.register_model( + model_cost={deployment_id: {"id": deployment_id, "access_groups": ["x"]}}, + persist_across_reloads=False, + ) + logging_obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}} + try: + assert litellm.get_model_info(model=deployment_id)["input_cost_per_token"] == 0 + assert logging_obj.get_router_deployment_model_info() is None + finally: + litellm.model_cost.pop(deployment_id, None) + def test_returns_none_without_a_deployment_id(self, logging_obj): logging_obj.litellm_params = {"api_base": ""} assert logging_obj.get_router_deployment_model_info() is None