fix(cost): stop non-string response service_tier from dropping cost tracking (#30706)

completion_cost extracted service_tier from the response object and the usage
object without an isinstance guard, so a non-string value (e.g. a dict) flowed
straight into _get_service_tier_cost_key and raised AttributeError on
service_tier.lower(). completion_cost re-raises, so the request's cost was lost.

PR #30690 fixed only the request-level optional_params path. This extends the
same guard to the response and usage paths by normalizing each extracted value:
a non-string tier (and the routing-only "auto" sentinel) is not billable, so it
coerces to None and pricing defers to the next concrete tier the provider served,
falling back to standard pricing when none is present.

Adds two regression tests driving a dict service_tier through completion_cost,
one on the response object (defers to the served usage tier) and one on the usage
object (prices at standard); both raise AttributeError before the fix.
This commit is contained in:
yuneng-jiang 2026-06-17 19:35:57 -07:00 • committed by GitHub
parent e568d8bffb
commit e122dac0db
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 121 additions and 9 deletions

View file

@ -888,6 +888,23 @@ def _map_traffic_type_to_service_tier(traffic_type: Optional[str]) -> Optional[s
return service_tier
def _normalize_service_tier(service_tier: object) -> str | None:
"""
Reduce a service_tier value to a concrete billable tier string or None.
"auto" is a routing preference and any non-string value is not a billable
tier, so both defer to standard pricing (or to the tier the provider reports
on the response usage) instead of crashing the downstream cost-key lookup,
which calls service_tier.lower()
"""
if (
not isinstance(service_tier, str)
or service_tier.lower() == ServiceTier.AUTO.value
):
return None
return service_tier
def _get_usage_object(
completion_response: Any,
) -> Optional[Usage]:
@ -1227,15 +1244,7 @@ def completion_cost(
if service_tier is None and optional_params is not None:
service_tier = optional_params.get("service_tier")
# A request-level service_tier only prices the request when it is a
# concrete billable tier string. "auto" is a routing preference and any
# non-string value is not a billable tier, so defer to the tier the
# provider reports on the response/usage instead of crashing or mispricing
if (
not isinstance(service_tier, str)
or service_tier.lower() == ServiceTier.AUTO.value
):
service_tier = None
service_tier = _normalize_service_tier(service_tier)
# Extract service_tier from completion_response if not provided
if service_tier is None and completion_response is not None:
@ -1244,6 +1253,8 @@ def completion_cost(
elif isinstance(completion_response, dict):
service_tier = completion_response.get("service_tier")
service_tier = _normalize_service_tier(service_tier)
# Extract service_tier from usage object if not provided
if service_tier is None and cost_per_token_usage_object is not None:
if isinstance(cost_per_token_usage_object, BaseModel):
@ -1253,6 +1264,8 @@ def completion_cost(
elif isinstance(cost_per_token_usage_object, dict):
service_tier = cost_per_token_usage_object.get("service_tier")
service_tier = _normalize_service_tier(service_tier)
selected_model = _select_model_name_for_cost_calc(
model=model,
completion_response=completion_response,

View file

@ -2285,6 +2285,105 @@ def test_completion_cost_non_string_service_tier_defers_to_served_tier():
assert cost == pytest.approx(expected_priority)
def test_completion_cost_non_string_response_service_tier_defers_to_served_tier():
"""
Regression: a non-string ``service_tier`` on the response object must not
crash cost tracking.
Before the fix ``completion_cost`` read the response-level value verbatim and
passed it to ``_get_service_tier_cost_key``, which called ``service_tier.lower()``
on the dict and raised ``AttributeError``. The non-string preference is not a
billable tier, so pricing defers to the concrete tier the provider served on
the usage object instead of crashing.
"""
from litellm import completion_cost
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-response-non-string-tier-cost-model"
litellm.register_model(
model_cost={
model: {
"input_cost_per_token": 3e-6,
"output_cost_per_token": 15e-6,
"input_cost_per_token_priority": 6e-6,
"output_cost_per_token_priority": 30e-6,
"litellm_provider": "anthropic",
"max_tokens": 8192,
}
}
)
usage = AnthropicConfig().calculate_usage(
usage_object={
"input_tokens": 1000,
"output_tokens": 500,
"service_tier": "priority",
},
reasoning_content=None,
)
response = ModelResponse(
usage=usage, model=model, service_tier={"name": "priority"}
)
cost = completion_cost(
completion_response=response,
model=model,
custom_llm_provider="anthropic",
)
expected_priority = 1000 * 6e-6 + 500 * 30e-6
assert cost == pytest.approx(expected_priority)
def test_completion_cost_non_string_usage_service_tier_prices_standard():
"""
Regression: a non-string ``service_tier`` on the usage object must not crash
cost tracking.
The dict reaches ``completion_cost`` via the usage extraction path with no
concrete tier to defer to, so pricing falls back to the standard rate instead
of raising ``AttributeError`` in ``_get_service_tier_cost_key``.
"""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-usage-non-string-tier-cost-model"
litellm.register_model(
model_cost={
model: {
"input_cost_per_token": 3e-6,
"output_cost_per_token": 15e-6,
"input_cost_per_token_priority": 6e-6,
"output_cost_per_token_priority": 30e-6,
"litellm_provider": "anthropic",
"max_tokens": 8192,
}
}
)
usage = Usage(
prompt_tokens=1000,
completion_tokens=500,
total_tokens=1500,
service_tier={"name": "priority"},
)
response = ModelResponse(usage=usage, model=model)
cost = completion_cost(
completion_response=response,
model=model,
custom_llm_provider="anthropic",
)
expected_standard = 1000 * 3e-6 + 500 * 15e-6
assert cost == pytest.approx(expected_standard)
def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier():
"""
Regression for the cache/tier interaction in the Anthropic geo/speed path.