mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(batches): only use deployment pricing when the deployment declares it
The router registers a model_info entry for every deployment, priced or not, and get_model_info fills absent costs with 0. Resolving deployment pricing through it therefore reported a free deployment for any ordinary one, which priced its batches at $0 while usage stayed correct: the same silent under-count this branch set out to remove, widened from bedrock to every provider. Caught by a live batch run, where four vertex batches that price correctly today came back at $0. The raw registration is now what decides: pricing is used only when the deployment actually declares one of the batch cost fields, so ordinary deployments fall back to the global cost map exactly as before. The earlier test missed this by using a deployment id that was never registered, where get_model_info does raise; a real deployment is always registered.
This commit is contained in:
parent
f74c72eedb
commit
727905dfc3
2 changed files with 32 additions and 2 deletions
|
|
@ -584,14 +584,26 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
"""Pricing the router registered under this deployment's model_info.id.
|
||||
|
||||
Returns None when the deployment declares no pricing of its own, so the
|
||||
caller falls back to the global cost map.
|
||||
caller falls back to the global cost map. The raw registration is what
|
||||
decides that: the router registers an entry for every deployment, and
|
||||
get_model_info fills absent costs with 0, so asking it directly cannot
|
||||
tell "configured as free" apart from "no pricing configured".
|
||||
"""
|
||||
pricing_keys: Final = (
|
||||
"input_cost_per_token",
|
||||
"output_cost_per_token",
|
||||
"input_cost_per_token_batches",
|
||||
"output_cost_per_token_batches",
|
||||
)
|
||||
model_id: Final = self.get_router_model_id()
|
||||
if model_id is None:
|
||||
return None
|
||||
registered: Final = litellm.model_cost.get(model_id)
|
||||
if not isinstance(registered, dict) or not any(registered.get(key) is not None for key in pricing_keys):
|
||||
return None
|
||||
try:
|
||||
return litellm.get_model_info(model=model_id)
|
||||
except Exception: # noqa: BLE001 # get_model_info raises for any id with no registered pricing
|
||||
except Exception: # noqa: BLE001 # get_model_info raises for ids it cannot resolve a provider for
|
||||
return None
|
||||
|
||||
def update_environment_variables(
|
||||
|
|
|
|||
|
|
@ -367,6 +367,24 @@ class TestGetRouterDeploymentModelInfo:
|
|||
logging_obj.litellm_params = {"litellm_metadata": {"model_info": {"id": "deploy-never-registered"}}}
|
||||
assert logging_obj.get_router_deployment_model_info() is None
|
||||
|
||||
def test_returns_none_when_deployment_registered_without_pricing(self, logging_obj):
|
||||
"""The router registers an entry for EVERY deployment, priced or not.
|
||||
|
||||
get_model_info fills absent costs with 0, so consulting it directly would
|
||||
hand back free pricing for an ordinary deployment and bill its batches $0.
|
||||
"""
|
||||
deployment_id = "deploy-no-pricing-1"
|
||||
litellm.register_model(
|
||||
model_cost={deployment_id: {"id": deployment_id, "access_groups": ["x"]}},
|
||||
persist_across_reloads=False,
|
||||
)
|
||||
logging_obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}}
|
||||
try:
|
||||
assert litellm.get_model_info(model=deployment_id)["input_cost_per_token"] == 0
|
||||
assert logging_obj.get_router_deployment_model_info() is None
|
||||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
||||
def test_returns_none_without_a_deployment_id(self, logging_obj):
|
||||
logging_obj.litellm_params = {"api_base": ""}
|
||||
assert logging_obj.get_router_deployment_model_info() is None
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue