From 76aa172e82149cfc97e4f8ad774ba9431303b222 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 5 Aug 2026 16:11:38 +0000 Subject: [PATCH] fix(cost): bedrock_mantle inherits bedrock regional pricing bedrock_mantle deployments fell back to the bare (US) price-map entry because no bedrock_mantle// key exists. Resolve pricing through the bedrock keys when bedrock_mantle has no entry of its own, so a mantle deployment bills the same regional rate as the equivalent bedrock deployment. Fixes #35953 --- litellm/cost_calculator.py | 22 +++++--- tests/test_litellm/test_cost_calculator.py | 62 ++++++++++++++++++++++ 2 files changed, 76 insertions(+), 8 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 025d400509b..237ff09cde1 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -450,14 +450,20 @@ def cost_per_token( model_with_provider = model if caller_supplied_provider: _prov_prefix: Final = f"{custom_llm_provider}/" - if model_is_str and model.startswith(_prov_prefix): - model_with_provider = model - else: - model_with_provider = f"{custom_llm_provider}/{model}" - if region_name is not None: - model_with_provider_and_region: Final = f"{custom_llm_provider}/{region_name}/{model}" - if model_with_provider_and_region in model_cost_ref: # use region based pricing, if it's available - model_with_provider = model_with_provider_and_region + _has_prov_prefix: Final = model_is_str and model.startswith(_prov_prefix) + _bare_model: Final = model[len(_prov_prefix) :] if _has_prov_prefix else model + _pricing_providers: Final[tuple[str, ...]] = ( + (custom_llm_provider, "bedrock") if custom_llm_provider == "bedrock_mantle" else (custom_llm_provider,) + ) + _regional_keys: Final[tuple[str, ...]] = ( + tuple(f"{_p}/{region_name}/{_bare_model}" for _p in _pricing_providers) if region_name is not None else () + ) + _flat_keys: Final[tuple[str, ...]] = tuple(f"{_p}/{_bare_model}" for _p in _pricing_providers) + _default_key: Final = model if _has_prov_prefix else f"{custom_llm_provider}/{model}" + model_with_provider = next( + (k for k in (*_regional_keys, *_flat_keys) if k in model_cost_ref), + _default_key, + ) else: _, custom_llm_provider, _, _ = litellm.get_llm_provider(model=model) diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index b22e16d8c6e..5e3ac4b64c6 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -42,6 +42,68 @@ def test_cost_per_token_duplicate_openai_prefix_matches_model_cost(monkeypatch): assert prompt_usd + completion_usd > 0 +def test_cost_per_token_bedrock_mantle_inherits_bedrock_regional_pricing(monkeypatch): + """ + A bedrock_mantle deployment has no price-map key of its own for models that + are also served over bedrock-runtime, so it must inherit the bedrock regional + rate instead of falling back to the bare (US) entry. Regression for #35953. + """ + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + eu_key = "bedrock/eu-north-1/moonshotai.kimi-k2.5" + us_key = "moonshotai.kimi-k2.5" + eu_in = litellm.model_cost[eu_key]["input_cost_per_token"] + eu_out = litellm.model_cost[eu_key]["output_cost_per_token"] + us_in = litellm.model_cost[us_key]["input_cost_per_token"] + assert eu_in != us_in # otherwise the test cannot distinguish the bug + + prompt_usd, completion_usd = cost_per_token( + model="moonshotai.kimi-k2.5", + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + custom_llm_provider="bedrock_mantle", + region_name="eu-north-1", + ) + + assert prompt_usd == pytest.approx(eu_in * 1_000_000) + assert completion_usd == pytest.approx(eu_out * 1_000_000) + + bedrock_prompt_usd, bedrock_completion_usd = cost_per_token( + model="moonshotai.kimi-k2.5", + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + custom_llm_provider="bedrock", + region_name="eu-north-1", + ) + assert (prompt_usd, completion_usd) == (bedrock_prompt_usd, bedrock_completion_usd) + + +def test_cost_per_token_bedrock_mantle_keeps_its_own_pricing(monkeypatch): + """ + The bedrock fallback must not override a model that has its own + bedrock_mantle price-map entry (e.g. mantle-exclusive gpt-oss). Guards the + #35953 fallback from becoming over-broad. + """ + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + mantle_key = "bedrock_mantle/openai.gpt-oss-120b" + mantle_in = litellm.model_cost[mantle_key]["input_cost_per_token"] + mantle_out = litellm.model_cost[mantle_key]["output_cost_per_token"] + + prompt_usd, completion_usd = cost_per_token( + model="openai.gpt-oss-120b", + prompt_tokens=1_000_000, + completion_tokens=1_000_000, + custom_llm_provider="bedrock_mantle", + region_name="eu-north-1", + ) + + assert prompt_usd == pytest.approx(mantle_in * 1_000_000) + assert completion_usd == pytest.approx(mantle_out * 1_000_000) + + def test_cost_per_token_non_string_model_does_not_hang(): """ The provider-prefix dedup loop must not spin forever when `model` is a