From 37102f76f5ce93786f7d20924bc6e9e4eedef063 Mon Sep 17 00:00:00 2001 From: Filippo Menghi <113345637+Cyberfilo@users.noreply.github.com> Date: Mon, 1 Jun 2026 13:02:55 +0200 Subject: [PATCH] Add 1-hour cache write pricing for us-gov Haiku 4.5 (#28574) * fix(thinking): handle None thinking param in is_thinking_enabled (#28598) Squash-merged by litellm-agent from Terrajlz's PR. * feat(helm): support tpl rendering in podAnnotations (#28609) Squash-merged by litellm-agent from devauxbr's PR. * Add 1-hour cache write pricing for us-gov Haiku 4.5 AWS Bedrock GovCloud applies a +20% premium over global Anthropic rates. Global Haiku 4.5 5m/1h cache write is $1.25 / $2.00 per MTok; us-gov is therefore $1.50 / $2.40 per MTok (the 5m rate was already correct in litellm; the 1h field was missing). Adds `cache_creation_input_token_cost_above_1hr: 2.4e-06` to the two us-gov Haiku 4.5 entries in both pricing JSON files: - bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0 - bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0 New parametrized regression test tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py pins both entries and enforces the 1.6x 5m-to-1h ratio invariant matching the pattern used by the existing Bedrock and Vertex 1h-cache tests. Companion to the us-gov Sonnet 4.5 pricing fix. * chore: trigger shin-agent re-eval on retargeted staging base * chore: trigger shin-agent re-eval against updated Greptile state --------- Co-authored-by: Terrajlz Co-authored-by: Bruno Devaux Co-authored-by: Sameer Kankute --- ...odel_prices_and_context_window_backup.json | 2 + model_prices_and_context_window.json | 2 + .../test_bedrock_usgov_haiku_1hr_cache.py | 47 +++++++++++++++++++ 3 files changed, 51 insertions(+) create mode 100644 tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 74c3d27e2ee..704127780be 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41420,6 +41420,7 @@ }, "bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.5e-06, + "cache_creation_input_token_cost_above_1hr": 2.4e-06, "cache_read_input_token_cost": 1.2e-07, "input_cost_per_token": 1.2e-06, "litellm_provider": "bedrock", @@ -41442,6 +41443,7 @@ }, "bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.5e-06, + "cache_creation_input_token_cost_above_1hr": 2.4e-06, "cache_read_input_token_cost": 1.2e-07, "input_cost_per_token": 1.2e-06, "litellm_provider": "bedrock", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 572c9b0ebd9..730494eab72 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41304,6 +41304,7 @@ }, "bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.5e-06, + "cache_creation_input_token_cost_above_1hr": 2.4e-06, "cache_read_input_token_cost": 1.2e-07, "input_cost_per_token": 1.2e-06, "litellm_provider": "bedrock", @@ -41326,6 +41327,7 @@ }, "bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.5e-06, + "cache_creation_input_token_cost_above_1hr": 2.4e-06, "cache_read_input_token_cost": 1.2e-07, "input_cost_per_token": 1.2e-06, "litellm_provider": "bedrock", diff --git a/tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py b/tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py new file mode 100644 index 00000000000..1312aa110d3 --- /dev/null +++ b/tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py @@ -0,0 +1,47 @@ +""" +Validate that AWS GovCloud (Bedrock us-gov-*) Haiku 4.5 entries carry +the 1-hour cache write tier. + +AWS Bedrock GovCloud pricing applies a +20% premium over global +Anthropic rates. Global Haiku 4.5 1h cache write is $2.00/MTok; us-gov +is therefore $2.40/MTok — exactly 1.6x the 5-minute rate of $1.50/MTok. + +Source: https://aws.amazon.com/bedrock/pricing/ +""" + +import json +import os + +import pytest + + +@pytest.fixture(scope="module") +def model_data(): + json_path = os.path.join( + os.path.dirname(__file__), "../../model_prices_and_context_window.json" + ) + with open(json_path) as f: + return json.load(f) + + +HAIKU_USGOV_KEYS = [ + "bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0", + "bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0", +] + + +@pytest.mark.parametrize("model_key", HAIKU_USGOV_KEYS) +def test_usgov_haiku_4_5_1hr_cache_write(model_data, model_key): + assert model_key in model_data, f"Missing model entry: {model_key}" + info = model_data[model_key] + assert ( + info["cache_creation_input_token_cost"] == 1.5e-06 + ), f"{model_key}: 5m cache write should be $1.50/MTok" + assert ( + info["cache_creation_input_token_cost_above_1hr"] == 2.4e-06 + ), f"{model_key}: 1h cache write should be $2.40/MTok" + ratio = ( + info["cache_creation_input_token_cost_above_1hr"] + / info["cache_creation_input_token_cost"] + ) + assert abs(ratio - 1.6) < 1e-9, f"{model_key}: 1h/5m ratio is {ratio}, expected 1.6"