fix(utils): resolve bedrock regional inference profiles to regional pricing in get_model_info (LIT-4056) (#32389)

* fix(utils): resolve bedrock regional inference profiles to regional pricing in get_model_info (LIT-4056)

* test(register_model): use a triple provider prefix as the unresolvable-key fixture

get_model_info now resolves bedrock/bedrock/... like a routing prefix, so the
double-prefix fixture stopped exercising the register_model fallback path.
Lock the new double-prefix resolution in as a model-info regression test
This commit is contained in:
Mateo Wang 2026-07-07 20:49:03 -07:00 • committed by GitHub
parent 4b0ac8b352
commit 734fd29e00
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 63 additions and 19 deletions

View file

@ -2619,8 +2619,9 @@ _CACHE_PRICING_FIELDS = (
def _resolve_builtin_model_cost_entry(key: str, provider: str) -> Optional[Dict[str, Any]]:
"""Best-effort lookup of a built-in ``model_cost`` entry for a custom key
whose shape ``get_model_info`` cannot resolve (double provider prefixes
like ``bedrock/bedrock/us.anthropic.claude-sonnet-4-6`` or region aliases).
whose shape ``get_model_info`` cannot resolve (repeated provider prefixes
like ``bedrock/bedrock/bedrock/us.anthropic.claude-sonnet-4-6`` or region
aliases).
Returns a copy of the matching entry so the caller can inherit its defaults
(most importantly cache pricing) without mutating the shared built-in.
@ -5052,9 +5053,9 @@ def _get_model_info_from_generalization(
candidates = [
potential_model_names["combined_model_name"],
model,
potential_model_names["split_model"],
potential_model_names["combined_stripped_model_name"],
potential_model_names["stripped_model_name"],
potential_model_names["split_model"],
]
for candidate in candidates:
generalized_info = match_fallback_generalization(candidate)
@ -5094,6 +5095,11 @@ def _get_potential_model_names(
stripped_model_name,
)
if custom_llm_provider in ("bedrock", "bedrock_converse"):
from litellm.llms.bedrock.common_utils import strip_bedrock_routing_prefix
split_model = strip_bedrock_routing_prefix(split_model)
return PotentialModelNamesAndCustomLLMProvider(
split_model=split_model,
combined_model_name=combined_model_name,
@ -5261,9 +5267,9 @@ def _get_model_info_helper(
Check if: (in order of specificity)
1. 'custom_llm_provider/model' in litellm.model_cost. Checks "groq/llama3-8b-8192" if model="llama3-8b-8192" and custom_llm_provider="groq"
2. 'model' in litellm.model_cost. Checks "gemini-1.5-pro-002" in litellm.model_cost if model="gemini-1.5-pro-002" and custom_llm_provider=None
3. 'combined_stripped_model_name' in litellm.model_cost. Checks if 'gemini/gemini-1.5-flash' in model map, if 'gemini/gemini-1.5-flash-001' given.
4. 'stripped_model_name' in litellm.model_cost. Checks if 'ft:gpt-3.5-turbo' in model map, if 'ft:gpt-3.5-turbo:my-org:custom_suffix:id' given.
5. 'split_model' in litellm.model_cost. Checks "llama3-8b-8192" in litellm.model_cost if model="groq/llama3-8b-8192"
3. 'split_model' in litellm.model_cost. Checks "au.anthropic.claude-opus-4-8" in litellm.model_cost if model="bedrock/au.anthropic.claude-opus-4-8"
4. 'combined_stripped_model_name' in litellm.model_cost. Checks if 'gemini/gemini-1.5-flash' in model map, if 'gemini/gemini-1.5-flash-001' given.
5. 'stripped_model_name' in litellm.model_cost. Checks if 'ft:gpt-3.5-turbo' in model map, if 'ft:gpt-3.5-turbo:my-org:custom_suffix:id' given.
"""
_model_info: Optional[Dict[str, Any]] = None
@ -5289,6 +5295,16 @@ def _get_model_info_helper(
custom_llm_provider=model_cost_custom_llm_provider,
):
_model_info = None
if _model_info is None:
_matched_key = _get_model_cost_key(split_model)
if _matched_key is not None:
key = _matched_key
_model_info = _get_model_info_from_model_cost(key=cast(str, key))
if not _check_provider_match(
model_info=_model_info,
custom_llm_provider=model_cost_custom_llm_provider,
):
_model_info = None
if _model_info is None:
_matched_key = _get_model_cost_key(combined_stripped_model_name)
if _matched_key is not None:
@ -5309,16 +5325,6 @@ def _get_model_info_helper(
custom_llm_provider=model_cost_custom_llm_provider,
):
_model_info = None
if _model_info is None:
_matched_key = _get_model_cost_key(split_model)
if _matched_key is not None:
key = _matched_key
_model_info = _get_model_info_from_model_cost(key=cast(str, key))
if not _check_provider_match(
model_info=_model_info,
custom_llm_provider=model_cost_custom_llm_provider,
):
_model_info = None
if _model_info is None:
generalization = _get_model_info_from_generalization(

View file

@ -320,8 +320,9 @@ def test_register_model_strips_none_litellm_provider_from_get_model_info(monkeyp
def test_register_model_inherits_builtin_cache_pricing_for_unmapped_key():
"""Registering a custom override under a key shape that
``get_model_info`` cannot resolve (e.g. a double provider prefix like
``bedrock/bedrock/us.anthropic.claude-sonnet-4-6``) must still inherit
``get_model_info`` cannot resolve (e.g. a triple provider prefix like
``bedrock/bedrock/bedrock/us.anthropic.claude-sonnet-4-6``; a double
prefix now resolves like a routing prefix) must still inherit
the built-in cache pricing for the underlying model.
Before the fix ``register_model`` fell back to an empty ``existing_model``
@ -341,7 +342,7 @@ def test_register_model_inherits_builtin_cache_pricing_for_unmapped_key():
litellm.model_cost = litellm.get_model_cost_map(url="")
builtin_key = "us.anthropic.claude-sonnet-4-6"
registered_key = f"bedrock/bedrock/{builtin_key}"
registered_key = f"bedrock/bedrock/bedrock/{builtin_key}"
builtin = litellm.model_cost[builtin_key]
assert builtin["cache_creation_input_token_cost"] > 0

View file

@ -1063,6 +1063,43 @@ def test_get_model_info_gemini():
assert info.get("rpm") is not None, f"{model} does not have rpm"
def test_get_model_info_bedrock_regional_inference_profile_pricing(local_model_cost_map):
"""Regression LIT-4056: with the bedrock/ routing prefix (plain, converse/, or
invoke/), the exact regional cost-map entry must win over the region-stripped
base entry, matching the unprefixed control form."""
regional = litellm.model_cost["au.anthropic.claude-opus-4-8"]
base = litellm.model_cost["anthropic.claude-opus-4-8"]
assert regional["input_cost_per_token"] > base["input_cost_per_token"]
for model in (
"bedrock/au.anthropic.claude-opus-4-8",
"bedrock/converse/au.anthropic.claude-opus-4-8",
"bedrock/invoke/au.anthropic.claude-opus-4-8",
):
info = litellm.get_model_info(model=model)
assert info["key"] == "au.anthropic.claude-opus-4-8", model
assert info["input_cost_per_token"] == regional["input_cost_per_token"], model
assert info["output_cost_per_token"] == regional["output_cost_per_token"], model
control = litellm.get_model_info(model="au.anthropic.claude-opus-4-8", custom_llm_provider="bedrock")
assert control["key"] == "au.anthropic.claude-opus-4-8"
def test_get_model_info_bedrock_regional_profile_without_entry_falls_back_to_base(local_model_cost_map):
"""A regional profile with no dedicated cost-map entry must still resolve to its
region-stripped base entry."""
assert "jp.anthropic.claude-opus-4-8" not in litellm.model_cost
info = litellm.get_model_info(model="bedrock/jp.anthropic.claude-opus-4-8")
assert info["key"] == "anthropic.claude-opus-4-8"
def test_get_model_info_bedrock_double_provider_prefix_resolves(local_model_cost_map):
"""A doubled bedrock/ prefix routes at runtime via strip_bedrock_routing_prefix,
so model info must resolve it to the same entry the request actually bills as."""
info = litellm.get_model_info(model="bedrock/bedrock/us.anthropic.claude-sonnet-4-6")
assert info["key"] == "us.anthropic.claude-sonnet-4-6"
def test_openai_models_in_model_info():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")