mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Apply us-gov 200k-tier GovCloud premium to cross-region sonnet-4-5
The us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0 cross-region inference profile carried _above_200k_tokens fields at the +10% commercial-US rate while the base tier was correctly at +20%. AWS GovCloud pricing applies the same +20% uplift across all tiers, so long-context requests through this profile were undercharging by ~9%. Updates four existing fields (input/output/cache_creation/cache_read _above_200k_tokens) and adds cache_creation_input_token_cost_above_1hr _above_200k_tokens for parity with the us. cross-region entry. Extends test_bedrock_usgov_pricing.py with five parametrized field assertions plus a 1.2x ratio invariant across all 200k-tier fields, so any future regression on either dimension is caught.
This commit is contained in:
parent
6e55e6f80e
commit
7a255ba896
3 changed files with 53 additions and 8 deletions
|
|
@ -31372,10 +31372,11 @@
|
|||
"cache_creation_input_token_cost_above_1hr": 7.2e-06,
|
||||
"cache_read_input_token_cost": 3.6e-07,
|
||||
"input_cost_per_token": 3.6e-06,
|
||||
"input_cost_per_token_above_200k_tokens": 6.6e-06,
|
||||
"output_cost_per_token_above_200k_tokens": 2.475e-05,
|
||||
"cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
|
||||
"input_cost_per_token_above_200k_tokens": 7.2e-06,
|
||||
"output_cost_per_token_above_200k_tokens": 2.7e-05,
|
||||
"cache_creation_input_token_cost_above_200k_tokens": 9.0e-06,
|
||||
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.44e-05,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 7.2e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 64000,
|
||||
|
|
|
|||
|
|
@ -31428,10 +31428,11 @@
|
|||
"cache_creation_input_token_cost_above_1hr": 7.2e-06,
|
||||
"cache_read_input_token_cost": 3.6e-07,
|
||||
"input_cost_per_token": 3.6e-06,
|
||||
"input_cost_per_token_above_200k_tokens": 6.6e-06,
|
||||
"output_cost_per_token_above_200k_tokens": 2.475e-05,
|
||||
"cache_creation_input_token_cost_above_200k_tokens": 8.25e-06,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 6.6e-07,
|
||||
"input_cost_per_token_above_200k_tokens": 7.2e-06,
|
||||
"output_cost_per_token_above_200k_tokens": 2.7e-05,
|
||||
"cache_creation_input_token_cost_above_200k_tokens": 9.0e-06,
|
||||
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.44e-05,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 7.2e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 64000,
|
||||
|
|
|
|||
|
|
@ -87,3 +87,46 @@ def test_usgov_carries_20_percent_premium_over_global(model_data):
|
|||
assert (
|
||||
abs(ratio - 1.2) < 1e-9
|
||||
), f"{field}: us-gov / global ratio is {ratio}, expected 1.2"
|
||||
|
||||
|
||||
# The us-gov.anthropic.* cross-region inference profile is the only us-gov
|
||||
# entry that carries the 1M-context `_above_200k_tokens` pricing tier — the
|
||||
# bedrock/us-gov-{east,west}-1/ entries are capped at 200k tokens.
|
||||
USGOV_CROSS_REGION_KEY = "us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0"
|
||||
|
||||
EXPECTED_USGOV_ABOVE_200K = {
|
||||
"input_cost_per_token_above_200k_tokens": 7.2e-06,
|
||||
"output_cost_per_token_above_200k_tokens": 2.7e-05,
|
||||
"cache_creation_input_token_cost_above_200k_tokens": 9.0e-06,
|
||||
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.44e-05,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 7.2e-07,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field,expected", EXPECTED_USGOV_ABOVE_200K.items())
|
||||
def test_usgov_cross_region_above_200k_carries_gov_premium(model_data, field, expected):
|
||||
"""The `_above_200k_tokens` tier on the us-gov cross-region inference
|
||||
profile must also carry the +20% GovCloud uplift. The original PR
|
||||
corrected the base rates but left the 200k-tier fields at the +10%
|
||||
commercial-US rates, undercharging long-context requests.
|
||||
"""
|
||||
info = model_data[USGOV_CROSS_REGION_KEY]
|
||||
assert field in info, f"{USGOV_CROSS_REGION_KEY}: missing field {field}"
|
||||
assert (
|
||||
info[field] == expected
|
||||
), f"{USGOV_CROSS_REGION_KEY}: {field} should be {expected} (got {info[field]})"
|
||||
|
||||
|
||||
def test_usgov_cross_region_above_200k_ratio_to_global(model_data):
|
||||
"""Cross-check via the property-based invariant: every `_above_200k_tokens`
|
||||
field on the us-gov cross-region profile must equal 1.2x the global
|
||||
anthropic.* rate, the same GovCloud uplift the base tier carries.
|
||||
"""
|
||||
global_key = "anthropic.claude-sonnet-4-5-20250929-v1:0"
|
||||
global_info = model_data[global_key]
|
||||
usgov_info = model_data[USGOV_CROSS_REGION_KEY]
|
||||
for field in EXPECTED_USGOV_ABOVE_200K:
|
||||
ratio = usgov_info[field] / global_info[field]
|
||||
assert (
|
||||
abs(ratio - 1.2) < 1e-9
|
||||
), f"{field}: us-gov / global ratio is {ratio}, expected 1.2"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue