Fix Bedrock service_tier cost propagation (#21172)

This commit is contained in:
Emerson Gomes 2026-02-16 22:30:10 -06:00 • committed by GitHub
parent 4978df8ebd
commit b67c140938
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 62 additions and 6 deletions

View file

@ -448,7 +448,9 @@ def cost_per_token( # noqa: PLR0915
elif custom_llm_provider == "anthropic":
return anthropic_cost_per_token(model=model, usage=usage_block)
elif custom_llm_provider == "bedrock":
return bedrock_cost_per_token(model=model, usage=usage_block)
return bedrock_cost_per_token(
model=model, usage=usage_block, service_tier=service_tier
)
elif custom_llm_provider == "openai":
return openai_cost_per_token(
model=model, usage=usage_block, service_tier=service_tier
@ -2146,4 +2148,3 @@ def handle_realtime_stream_cost_calculation(
return total_cost

View file

@ -3,7 +3,7 @@ Helper util for handling bedrock-specific cost calculation
- e.g.: prompt caching
"""
from typing import TYPE_CHECKING, Tuple
from typing import TYPE_CHECKING, Optional, Tuple
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
@ -11,12 +11,17 @@ if TYPE_CHECKING:
from litellm.types.utils import Usage
def cost_per_token(model: str, usage: "Usage") -> Tuple[float, float]:
def cost_per_token(
model: str, usage: "Usage", service_tier: Optional[str] = None
) -> Tuple[float, float]:
"""
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
Follows the same logic as Anthropic's cost per token calculation.
"""
return generic_cost_per_token(
model=model, usage=usage, custom_llm_provider="bedrock"
)
model=model,
usage=usage,
custom_llm_provider="bedrock",
service_tier=service_tier,
)

View file

@ -1600,6 +1600,56 @@ def test_completion_cost_service_tier_priority():
), "Costs from params and usage should be similar (both flex)"
def test_completion_cost_service_tier_for_bedrock():
"""Test that Bedrock cost calculation applies service_tier-specific pricing."""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "bedrock/us-east-1/test-bedrock-service-tier-cost-model"
litellm.register_model(
model_cost={
model: {
"input_cost_per_token": 0.001,
"output_cost_per_token": 0.002,
"input_cost_per_token_priority": 0.01,
"output_cost_per_token_priority": 0.02,
"input_cost_per_token_flex": 0.0005,
"output_cost_per_token_flex": 0.001,
"litellm_provider": "bedrock",
"max_tokens": 8192,
}
}
)
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
response = ModelResponse(usage=usage, model=model)
default_cost = completion_cost(
completion_response=response,
model=model,
custom_llm_provider="bedrock",
)
priority_cost = completion_cost(
completion_response=response,
model=model,
custom_llm_provider="bedrock",
optional_params={"service_tier": "priority"},
)
response_with_flex_tier = ModelResponse(usage=usage, model=model)
setattr(response_with_flex_tier, "service_tier", "flex")
flex_cost = completion_cost(
completion_response=response_with_flex_tier,
model=model,
custom_llm_provider="bedrock",
)
assert priority_cost > default_cost > flex_cost > 0
def test_gemini_cache_tokens_details_no_negative_values():
"""
Test for Issue #18750: Negative text_tokens with Gemini caching