fix(cost): bedrock_mantle inherits bedrock regional pricing

bedrock_mantle deployments fell back to the bare (US) price-map entry
because no bedrock_mantle/<region>/<model> key exists. Resolve pricing
through the bedrock keys when bedrock_mantle has no entry of its own, so
a mantle deployment bills the same regional rate as the equivalent
bedrock deployment.

Fixes #35953
This commit is contained in:
devin-ai-integration[bot] 2026-08-05 16:11:38 +00:00 committed by GitHub
parent b735578822
commit 76aa172e82
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 76 additions and 8 deletions

View file

@ -450,14 +450,20 @@ def cost_per_token(
model_with_provider = model
if caller_supplied_provider:
_prov_prefix: Final = f"{custom_llm_provider}/"
if model_is_str and model.startswith(_prov_prefix):
model_with_provider = model
else:
model_with_provider = f"{custom_llm_provider}/{model}"
if region_name is not None:
model_with_provider_and_region: Final = f"{custom_llm_provider}/{region_name}/{model}"
if model_with_provider_and_region in model_cost_ref: # use region based pricing, if it's available
model_with_provider = model_with_provider_and_region
_has_prov_prefix: Final = model_is_str and model.startswith(_prov_prefix)
_bare_model: Final = model[len(_prov_prefix) :] if _has_prov_prefix else model
_pricing_providers: Final[tuple[str, ...]] = (
(custom_llm_provider, "bedrock") if custom_llm_provider == "bedrock_mantle" else (custom_llm_provider,)
)
_regional_keys: Final[tuple[str, ...]] = (
tuple(f"{_p}/{region_name}/{_bare_model}" for _p in _pricing_providers) if region_name is not None else ()
)
_flat_keys: Final[tuple[str, ...]] = tuple(f"{_p}/{_bare_model}" for _p in _pricing_providers)
_default_key: Final = model if _has_prov_prefix else f"{custom_llm_provider}/{model}"
model_with_provider = next(
(k for k in (*_regional_keys, *_flat_keys) if k in model_cost_ref),
_default_key,
)
else:
_, custom_llm_provider, _, _ = litellm.get_llm_provider(model=model)

View file

@ -42,6 +42,68 @@ def test_cost_per_token_duplicate_openai_prefix_matches_model_cost(monkeypatch):
assert prompt_usd + completion_usd > 0
def test_cost_per_token_bedrock_mantle_inherits_bedrock_regional_pricing(monkeypatch):
"""
A bedrock_mantle deployment has no price-map key of its own for models that
are also served over bedrock-runtime, so it must inherit the bedrock regional
rate instead of falling back to the bare (US) entry. Regression for #35953.
"""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
eu_key = "bedrock/eu-north-1/moonshotai.kimi-k2.5"
us_key = "moonshotai.kimi-k2.5"
eu_in = litellm.model_cost[eu_key]["input_cost_per_token"]
eu_out = litellm.model_cost[eu_key]["output_cost_per_token"]
us_in = litellm.model_cost[us_key]["input_cost_per_token"]
assert eu_in != us_in # otherwise the test cannot distinguish the bug
prompt_usd, completion_usd = cost_per_token(
model="moonshotai.kimi-k2.5",
prompt_tokens=1_000_000,
completion_tokens=1_000_000,
custom_llm_provider="bedrock_mantle",
region_name="eu-north-1",
)
assert prompt_usd == pytest.approx(eu_in * 1_000_000)
assert completion_usd == pytest.approx(eu_out * 1_000_000)
bedrock_prompt_usd, bedrock_completion_usd = cost_per_token(
model="moonshotai.kimi-k2.5",
prompt_tokens=1_000_000,
completion_tokens=1_000_000,
custom_llm_provider="bedrock",
region_name="eu-north-1",
)
assert (prompt_usd, completion_usd) == (bedrock_prompt_usd, bedrock_completion_usd)
def test_cost_per_token_bedrock_mantle_keeps_its_own_pricing(monkeypatch):
"""
The bedrock fallback must not override a model that has its own
bedrock_mantle price-map entry (e.g. mantle-exclusive gpt-oss). Guards the
#35953 fallback from becoming over-broad.
"""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
mantle_key = "bedrock_mantle/openai.gpt-oss-120b"
mantle_in = litellm.model_cost[mantle_key]["input_cost_per_token"]
mantle_out = litellm.model_cost[mantle_key]["output_cost_per_token"]
prompt_usd, completion_usd = cost_per_token(
model="openai.gpt-oss-120b",
prompt_tokens=1_000_000,
completion_tokens=1_000_000,
custom_llm_provider="bedrock_mantle",
region_name="eu-north-1",
)
assert prompt_usd == pytest.approx(mantle_in * 1_000_000)
assert completion_usd == pytest.approx(mantle_out * 1_000_000)
def test_cost_per_token_non_string_model_does_not_hang():
"""
The provider-prefix dedup loop must not spin forever when `model` is a