From a54e9062a64d9abf4bf986a83f040858134080df Mon Sep 17 00:00:00 2001 From: Bharat Lakhiyani Date: Fri, 24 Jul 2026 23:18:01 -0400 Subject: [PATCH 1/2] feat(bedrock_mantle): add Claude Opus 5, Sonnet 5, and Haiku 4.5 support The Mantle endpoint serves Claude models under dateless IDs (anthropic.claude-opus-5, anthropic.claude-sonnet-5, anthropic.claude-haiku-4-5 per the AWS model cards). Two gaps prevented correct handling: 1. anthropic.claude-haiku-4-5 had no cost-map entry (only the dated bedrock-runtime ID anthropic.claude-haiku-4-5-20251001-v1:0 exists), so Mantle Haiku responses were cost-tracked at $0. Add the dateless entry (200K in / 64K out, $1/$5 per 1M tokens, prompt caching). 2. strip_bedrock_routing_prefix did not strip the mantle/ routing prefix, so get_model_info on bedrock/mantle/anthropic.claude-* resolved generalization fallbacks (generic 200K/64K limits, no cache pricing, no output_config capability flags) instead of the exact bare-model entries. Add mantle/ to the prefix list; route detection is unaffected because explicit route prefixes are matched before base-model resolution, and the bedrock_mantle/ provider prefix is still left intact (segment- boundary matching). --- litellm/llms/bedrock/common_utils.py | 2 +- ...odel_prices_and_context_window_backup.json | 25 ++++++ model_prices_and_context_window.json | 25 ++++++ .../test_litellm/llms/bedrock/test_mantle.py | 88 +++++++++++++++++++ 4 files changed, 139 insertions(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 5114677ffc0..46afdcf7398 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -604,7 +604,7 @@ def is_bedrock_application_inference_profile_arn(model: str) -> bool: def strip_bedrock_routing_prefix(model: str) -> str: """Strip LiteLLM routing prefixes from model name.""" - for prefix in ["bedrock/", "converse/", "invoke/", "openai/", "nova-2/", "nova/"]: + for prefix in ["bedrock/", "converse/", "invoke/", "openai/", "mantle/", "nova-2/", "nova/"]: if model.startswith(prefix): model = model.split("/", 1)[1] return model diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d43eda39b1f..5d8d475739b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -724,6 +724,31 @@ "supports_tool_choice": true, "prompt_cache_min_tokens": 2048 }, + "anthropic.claude-haiku-4-5": { + "cache_creation_input_token_cost": 1.25e-06, + "cache_creation_input_token_cost_above_1hr": 2e-06, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 1e-06, + "litellm_provider": "bedrock", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 5e-06, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-4-5.html", + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_native_structured_output": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 4096 + }, "anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.25e-06, "cache_creation_input_token_cost_above_1hr": 2e-06, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 749b2566c2a..a802209ba5c 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -724,6 +724,31 @@ "supports_tool_choice": true, "prompt_cache_min_tokens": 2048 }, + "anthropic.claude-haiku-4-5": { + "cache_creation_input_token_cost": 1.25e-06, + "cache_creation_input_token_cost_above_1hr": 2e-06, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 1e-06, + "litellm_provider": "bedrock", + "max_input_tokens": 200000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 5e-06, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-4-5.html", + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_native_structured_output": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 4096 + }, "anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.25e-06, "cache_creation_input_token_cost_above_1hr": 2e-06, diff --git a/tests/test_litellm/llms/bedrock/test_mantle.py b/tests/test_litellm/llms/bedrock/test_mantle.py index d34517f61f6..cce00a01a9d 100644 --- a/tests/test_litellm/llms/bedrock/test_mantle.py +++ b/tests/test_litellm/llms/bedrock/test_mantle.py @@ -687,3 +687,91 @@ async def test_mantle_anthropic_messages_streaming_sends_stream_and_passes_throu assert "event: message_start" in text assert '"text": "pong"' in text assert "event: message_stop" in text + + +@pytest.fixture +def local_cost_map(monkeypatch): + import litellm + + original_model_cost = litellm.model_cost + try: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") + litellm.model_cost = litellm.get_model_cost_map(url="") + litellm.get_model_info.cache_clear() + yield + finally: + litellm.model_cost = original_model_cost + litellm.get_model_info.cache_clear() + + +@pytest.mark.parametrize( + "model_id,max_input,max_output,input_cost,output_cost", + [ + ("anthropic.claude-opus-5", 1_000_000, 128_000, 5e-06, 2.5e-05), + ("anthropic.claude-sonnet-5", 1_000_000, 128_000, 2e-06, 1e-05), + ("anthropic.claude-haiku-4-5", 200_000, 64_000, 1e-06, 5e-06), + ], +) +def test_mantle_route_resolves_exact_model_info( + local_cost_map, model_id, max_input, max_output, input_cost, output_cost +): + """The mantle/ routing prefix must resolve the exact bare-model cost-map + entry, not a generalization fallback. Before the fix, + strip_bedrock_routing_prefix left the mantle/ prefix in place, so + bedrock/mantle/anthropic.claude-opus-5 resolved generic 200k/64k limits + instead of the 1M/128k entry, and anthropic.claude-haiku-4-5 (the dateless + ID the Mantle endpoint uses) had no entry at all and cost $0.""" + import litellm + + info = litellm.get_model_info(f"bedrock/mantle/{model_id}") + assert info["max_input_tokens"] == max_input + assert info["max_output_tokens"] == max_output + assert info["input_cost_per_token"] == input_cost + assert info["output_cost_per_token"] == output_cost + + +@pytest.mark.parametrize( + "model_id", + [ + "anthropic.claude-opus-5", + "anthropic.claude-sonnet-5", + "anthropic.claude-haiku-4-5", + ], +) +def test_get_bedrock_route_mantle_claude_models(model_id): + assert BedrockModelInfo.get_bedrock_route(f"mantle/{model_id}") == "mantle" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/mantle/{model_id}") == "mantle" + + +def test_strip_bedrock_routing_prefix_strips_mantle_but_not_provider_prefix(): + from litellm.llms.bedrock.common_utils import strip_bedrock_routing_prefix + + assert ( + strip_bedrock_routing_prefix("bedrock/mantle/anthropic.claude-opus-5") + == "anthropic.claude-opus-5" + ) + assert ( + strip_bedrock_routing_prefix("mantle/anthropic.claude-sonnet-5") + == "anthropic.claude-sonnet-5" + ) + assert ( + strip_bedrock_routing_prefix("bedrock_mantle/openai.gpt-5.5") + == "bedrock_mantle/openai.gpt-5.5" + ) + + +def test_mantle_haiku_4_5_cost_tracking(local_cost_map): + """Regression: the Mantle endpoint serves Haiku 4.5 under the dateless ID + anthropic.claude-haiku-4-5 (bedrock-runtime uses the dated + anthropic.claude-haiku-4-5-20251001-v1:0). Without a dateless entry, + completion_cost silently returned 0.0.""" + import litellm + from litellm.types.utils import Choices, Message, ModelResponse, Usage + + response = ModelResponse( + model="anthropic.claude-haiku-4-5", + choices=[Choices(message=Message(content="ok"))], + usage=Usage(prompt_tokens=1000, completion_tokens=1000), + ) + cost = litellm.completion_cost(completion_response=response, custom_llm_provider="bedrock") + assert cost == pytest.approx(1000 * 1e-06 + 1000 * 5e-06) From b2567531f709d5fd53283135f6aea714246c0bfe Mon Sep 17 00:00:00 2001 From: Bharat Lakhiyani Date: Sat, 25 Jul 2026 00:06:31 -0400 Subject: [PATCH 2/2] test(bedrock): clear all model-cost LRU caches in local_cost_map fixture get_model_info.cache_clear() alone leaves _cached_get_model_info_helper (used by completion_cost) holding entries from whichever cost map was active first, making the cost-tracking test order-dependent. Use _invalidate_model_cost_lowercase_map, which clears both caches and the lowercase lookup map. --- tests/test_litellm/llms/bedrock/test_mantle.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/tests/test_litellm/llms/bedrock/test_mantle.py b/tests/test_litellm/llms/bedrock/test_mantle.py index cce00a01a9d..c219cc8a47d 100644 --- a/tests/test_litellm/llms/bedrock/test_mantle.py +++ b/tests/test_litellm/llms/bedrock/test_mantle.py @@ -692,16 +692,17 @@ async def test_mantle_anthropic_messages_streaming_sends_stream_and_passes_throu @pytest.fixture def local_cost_map(monkeypatch): import litellm + from litellm.utils import _invalidate_model_cost_lowercase_map original_model_cost = litellm.model_cost try: monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") litellm.model_cost = litellm.get_model_cost_map(url="") - litellm.get_model_info.cache_clear() + _invalidate_model_cost_lowercase_map() yield finally: litellm.model_cost = original_model_cost - litellm.get_model_info.cache_clear() + _invalidate_model_cost_lowercase_map() @pytest.mark.parametrize(