From 7a255ba896012034fed4d8d931f0fa596e933a71 Mon Sep 17 00:00:00 2001 From: Filippo Mattia Menghi Date: Mon, 25 May 2026 13:28:38 +0200 Subject: [PATCH] Apply us-gov 200k-tier GovCloud premium to cross-region sonnet-4-5 The us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0 cross-region inference profile carried _above_200k_tokens fields at the +10% commercial-US rate while the base tier was correctly at +20%. AWS GovCloud pricing applies the same +20% uplift across all tiers, so long-context requests through this profile were undercharging by ~9%. Updates four existing fields (input/output/cache_creation/cache_read _above_200k_tokens) and adds cache_creation_input_token_cost_above_1hr _above_200k_tokens for parity with the us. cross-region entry. Extends test_bedrock_usgov_pricing.py with five parametrized field assertions plus a 1.2x ratio invariant across all 200k-tier fields, so any future regression on either dimension is caught. --- ...odel_prices_and_context_window_backup.json | 9 ++-- model_prices_and_context_window.json | 9 ++-- .../test_bedrock_usgov_pricing.py | 43 +++++++++++++++++++ 3 files changed, 53 insertions(+), 8 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ba5bc13dac3..e9404da8a85 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -31372,10 +31372,11 @@ "cache_creation_input_token_cost_above_1hr": 7.2e-06, "cache_read_input_token_cost": 3.6e-07, "input_cost_per_token": 3.6e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, - "output_cost_per_token_above_200k_tokens": 2.475e-05, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, + "input_cost_per_token_above_200k_tokens": 7.2e-06, + "output_cost_per_token_above_200k_tokens": 2.7e-05, + "cache_creation_input_token_cost_above_200k_tokens": 9.0e-06, + "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.44e-05, + "cache_read_input_token_cost_above_200k_tokens": 7.2e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 200000, "max_output_tokens": 64000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 705c2971dca..1b6d86bc7e1 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -31428,10 +31428,11 @@ "cache_creation_input_token_cost_above_1hr": 7.2e-06, "cache_read_input_token_cost": 3.6e-07, "input_cost_per_token": 3.6e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, - "output_cost_per_token_above_200k_tokens": 2.475e-05, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, + "input_cost_per_token_above_200k_tokens": 7.2e-06, + "output_cost_per_token_above_200k_tokens": 2.7e-05, + "cache_creation_input_token_cost_above_200k_tokens": 9.0e-06, + "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.44e-05, + "cache_read_input_token_cost_above_200k_tokens": 7.2e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 200000, "max_output_tokens": 64000, diff --git a/tests/test_litellm/test_bedrock_usgov_pricing.py b/tests/test_litellm/test_bedrock_usgov_pricing.py index 03ae25858a8..6b3312b5cc4 100644 --- a/tests/test_litellm/test_bedrock_usgov_pricing.py +++ b/tests/test_litellm/test_bedrock_usgov_pricing.py @@ -87,3 +87,46 @@ def test_usgov_carries_20_percent_premium_over_global(model_data): assert ( abs(ratio - 1.2) < 1e-9 ), f"{field}: us-gov / global ratio is {ratio}, expected 1.2" + + +# The us-gov.anthropic.* cross-region inference profile is the only us-gov +# entry that carries the 1M-context `_above_200k_tokens` pricing tier — the +# bedrock/us-gov-{east,west}-1/ entries are capped at 200k tokens. +USGOV_CROSS_REGION_KEY = "us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0" + +EXPECTED_USGOV_ABOVE_200K = { + "input_cost_per_token_above_200k_tokens": 7.2e-06, + "output_cost_per_token_above_200k_tokens": 2.7e-05, + "cache_creation_input_token_cost_above_200k_tokens": 9.0e-06, + "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.44e-05, + "cache_read_input_token_cost_above_200k_tokens": 7.2e-07, +} + + +@pytest.mark.parametrize("field,expected", EXPECTED_USGOV_ABOVE_200K.items()) +def test_usgov_cross_region_above_200k_carries_gov_premium(model_data, field, expected): + """The `_above_200k_tokens` tier on the us-gov cross-region inference + profile must also carry the +20% GovCloud uplift. The original PR + corrected the base rates but left the 200k-tier fields at the +10% + commercial-US rates, undercharging long-context requests. + """ + info = model_data[USGOV_CROSS_REGION_KEY] + assert field in info, f"{USGOV_CROSS_REGION_KEY}: missing field {field}" + assert ( + info[field] == expected + ), f"{USGOV_CROSS_REGION_KEY}: {field} should be {expected} (got {info[field]})" + + +def test_usgov_cross_region_above_200k_ratio_to_global(model_data): + """Cross-check via the property-based invariant: every `_above_200k_tokens` + field on the us-gov cross-region profile must equal 1.2x the global + anthropic.* rate, the same GovCloud uplift the base tier carries. + """ + global_key = "anthropic.claude-sonnet-4-5-20250929-v1:0" + global_info = model_data[global_key] + usgov_info = model_data[USGOV_CROSS_REGION_KEY] + for field in EXPECTED_USGOV_ABOVE_200K: + ratio = usgov_info[field] / global_info[field] + assert ( + abs(ratio - 1.2) < 1e-9 + ), f"{field}: us-gov / global ratio is {ratio}, expected 1.2"