mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Add 1-hour cache write pricing for us-gov Haiku 4.5 (#28574)
* fix(thinking): handle None thinking param in is_thinking_enabled (#28598) Squash-merged by litellm-agent from Terrajlz's PR. * feat(helm): support tpl rendering in podAnnotations (#28609) Squash-merged by litellm-agent from devauxbr's PR. * Add 1-hour cache write pricing for us-gov Haiku 4.5 AWS Bedrock GovCloud applies a +20% premium over global Anthropic rates. Global Haiku 4.5 5m/1h cache write is $1.25 / $2.00 per MTok; us-gov is therefore $1.50 / $2.40 per MTok (the 5m rate was already correct in litellm; the 1h field was missing). Adds `cache_creation_input_token_cost_above_1hr: 2.4e-06` to the two us-gov Haiku 4.5 entries in both pricing JSON files: - bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0 - bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0 New parametrized regression test tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py pins both entries and enforces the 1.6x 5m-to-1h ratio invariant matching the pattern used by the existing Bedrock and Vertex 1h-cache tests. Companion to the us-gov Sonnet 4.5 pricing fix. * chore: trigger shin-agent re-eval on retargeted staging base * chore: trigger shin-agent re-eval against updated Greptile state --------- Co-authored-by: Terrajlz <info@jouleselectrictech.com> Co-authored-by: Bruno Devaux <devaux.br@gmail.com> Co-authored-by: Sameer Kankute <sameer@berri.ai>
This commit is contained in:
parent
fd554501ef
commit
37102f76f5
3 changed files with 51 additions and 0 deletions
|
|
@ -41420,6 +41420,7 @@
|
|||
},
|
||||
"bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0": {
|
||||
"cache_creation_input_token_cost": 1.5e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -41442,6 +41443,7 @@
|
|||
},
|
||||
"bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0": {
|
||||
"cache_creation_input_token_cost": 1.5e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
|
|||
|
|
@ -41304,6 +41304,7 @@
|
|||
},
|
||||
"bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0": {
|
||||
"cache_creation_input_token_cost": 1.5e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
@ -41326,6 +41327,7 @@
|
|||
},
|
||||
"bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0": {
|
||||
"cache_creation_input_token_cost": 1.5e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.4e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
"litellm_provider": "bedrock",
|
||||
|
|
|
|||
47
tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py
Normal file
47
tests/test_litellm/test_bedrock_usgov_haiku_1hr_cache.py
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
"""
|
||||
Validate that AWS GovCloud (Bedrock us-gov-*) Haiku 4.5 entries carry
|
||||
the 1-hour cache write tier.
|
||||
|
||||
AWS Bedrock GovCloud pricing applies a +20% premium over global
|
||||
Anthropic rates. Global Haiku 4.5 1h cache write is $2.00/MTok; us-gov
|
||||
is therefore $2.40/MTok — exactly 1.6x the 5-minute rate of $1.50/MTok.
|
||||
|
||||
Source: https://aws.amazon.com/bedrock/pricing/
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def model_data():
|
||||
json_path = os.path.join(
|
||||
os.path.dirname(__file__), "../../model_prices_and_context_window.json"
|
||||
)
|
||||
with open(json_path) as f:
|
||||
return json.load(f)
|
||||
|
||||
|
||||
HAIKU_USGOV_KEYS = [
|
||||
"bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0",
|
||||
"bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0",
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model_key", HAIKU_USGOV_KEYS)
|
||||
def test_usgov_haiku_4_5_1hr_cache_write(model_data, model_key):
|
||||
assert model_key in model_data, f"Missing model entry: {model_key}"
|
||||
info = model_data[model_key]
|
||||
assert (
|
||||
info["cache_creation_input_token_cost"] == 1.5e-06
|
||||
), f"{model_key}: 5m cache write should be $1.50/MTok"
|
||||
assert (
|
||||
info["cache_creation_input_token_cost_above_1hr"] == 2.4e-06
|
||||
), f"{model_key}: 1h cache write should be $2.40/MTok"
|
||||
ratio = (
|
||||
info["cache_creation_input_token_cost_above_1hr"]
|
||||
/ info["cache_creation_input_token_cost"]
|
||||
)
|
||||
assert abs(ratio - 1.6) < 1e-9, f"{model_key}: 1h/5m ratio is {ratio}, expected 1.6"
|
||||
Loading…
Add table
Reference in a new issue