mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(bedrock): stop emitting Converse cachePoint blocks for Kimi K3 (#44292)
* fix(bedrock): stop emitting Converse cachePoint blocks for Kimi K3
Bedrock prices Kimi K3 cache reads through implicit caching but Converse
rejects the explicit cachePoint marker ("This model doesn't support the
cachePoint field"), so any cache_control on the request answered 400.
Mark the three K3 rows supports_prompt_cache_breakpoint: false and have
bedrock_model_accepts_cache_points honor that flag before falling back to
supports_prompt_caching, keeping cached-token pricing intact.
* test(bedrock): assert cache points per request section
* fix(bedrock): honor a deployment's cache breakpoint flag for unmapped models
* fix(bedrock): read a converse-routed deployment's cache breakpoint flag
* refactor(bedrock): look up cache breakpoint flags by key
* test(bedrock): add the Kimi K3 cache point wire audit
---------
Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
4732647de2
commit
6c32384d8c
6 changed files with 1699 additions and 17 deletions
|
|
@ -1125,23 +1125,42 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
|
|||
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
|
||||
models without prompt caching support ("You invoked an unsupported model or your
|
||||
request did not allow prompt caching"), so a model whose cost-map entry does not declare
|
||||
``supports_prompt_caching`` must not receive them. A model absent from the map
|
||||
(an application inference profile ARN, a model newer than the map) keeps emitting
|
||||
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
|
||||
is not reusable here: it returns False for unmapped models, the opposite polarity.
|
||||
``supports_prompt_caching`` must not receive them. An explicit
|
||||
``supports_prompt_cache_breakpoint`` on the entry wins over that flag: a model can price
|
||||
cached tokens through implicit caching yet reject the marker on Converse ("This model
|
||||
doesn't support the cachePoint field", Kimi K3). The router registers a deployment's
|
||||
``model_info`` under ``bedrock/<model>`` as configured, route prefix included, while the
|
||||
Converse transformation sees the model with ``converse/`` or ``converse_like/`` already
|
||||
stripped, so every registration form is read. That flag set there covers an application
|
||||
inference profile ARN or a model newer than the map, while only the map decides whether
|
||||
a model is known: absent a map entry the model keeps emitting so existing caching setups
|
||||
never silently degrade. ``litellm.utils.supports_prompt_caching`` is not reusable here:
|
||||
it returns False for unmapped models, the opposite polarity.
|
||||
"""
|
||||
if model is None:
|
||||
return True
|
||||
if _OPENAI_FAMILY_MODEL_RE.search(model):
|
||||
return False
|
||||
entries: Final = tuple(
|
||||
entry
|
||||
for candidate in (model, get_bedrock_base_model(model))
|
||||
if (entry := litellm.model_cost.get(candidate)) is not None
|
||||
map_keys: Final = (model, get_bedrock_base_model(model))
|
||||
registered_keys: Final = tuple(f"bedrock/{route}{model}" for route in ("", "converse/", "converse_like/"))
|
||||
explicit_marker_support: Final = next(
|
||||
(
|
||||
entry.get("supports_prompt_cache_breakpoint") is True
|
||||
for key in (*registered_keys, *map_keys)
|
||||
if (entry := litellm.model_cost.get(key)) is not None
|
||||
and entry.get("supports_prompt_cache_breakpoint") is not None
|
||||
),
|
||||
None,
|
||||
)
|
||||
if not entries:
|
||||
if explicit_marker_support is not None:
|
||||
return explicit_marker_support
|
||||
if not any(key in litellm.model_cost for key in map_keys):
|
||||
return True
|
||||
return any(entry.get("supports_prompt_caching") is True for entry in entries)
|
||||
return any(
|
||||
entry.get("supports_prompt_caching") is True
|
||||
for key in map_keys
|
||||
if (entry := litellm.model_cost.get(key)) is not None
|
||||
)
|
||||
|
||||
|
||||
def bedrock_supports_tool_search(model: str) -> bool:
|
||||
|
|
|
|||
|
|
@ -76931,6 +76931,7 @@
|
|||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_cache_breakpoint": false,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
|
|
@ -76952,6 +76953,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_cache_breakpoint": false,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
|
|
@ -76973,6 +76975,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_cache_breakpoint": false,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
|
|
|
|||
|
|
@ -76931,6 +76931,7 @@
|
|||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_cache_breakpoint": false,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
|
|
@ -76952,6 +76953,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_cache_breakpoint": false,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
|
|
@ -76973,6 +76975,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_cache_breakpoint": false,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
|
|
|
|||
1572
tests/integration/providers/test_bedrock_kimi_k3_cache_point_wire.py
Normal file
1572
tests/integration/providers/test_bedrock_kimi_k3_cache_point_wire.py
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -5769,15 +5769,20 @@ def test_cache_control_injection_tool_config_drops_ttl_for_unsupported_model():
|
|||
pytest.param("global.openai.gpt-6-astra", False, id="openai-family-implicit-caching-only"),
|
||||
pytest.param("openai.gpt-oss-120b-1:0", False, id="openai-gpt-oss"),
|
||||
pytest.param("us.openai.gpt-99-unmapped", False, id="unmapped-openai-family-still-suppressed"),
|
||||
pytest.param("us.moonshotai.kimi-k3", False, id="kimi-k3-prices-cached-tokens-but-rejects-cachepoint"),
|
||||
pytest.param("global.moonshotai.kimi-k3", False, id="kimi-k3-global-profile"),
|
||||
pytest.param("us-east-1/us.moonshotai.kimi-k3", False, id="kimi-k3-regional-route-resolves-through-profile"),
|
||||
],
|
||||
)
|
||||
def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model, expects_cache_points, monkeypatch):
|
||||
"""Bedrock rejects cachePoint blocks for models without prompt caching support
|
||||
("You invoked an unsupported model or your request did not allow prompt caching"),
|
||||
and clients like Claude Code attach cache_control to every request, so a map-known
|
||||
model without the capability must not receive them. Unmapped ids (application
|
||||
inference profile ARNs, models newer than the map) keep emitting so existing
|
||||
caching setups never silently degrade."""
|
||||
("You invoked an unsupported model or your request did not allow prompt caching")
|
||||
and for models that price cached tokens yet take the marker only on their native
|
||||
endpoints ("This model doesn't support the cachePoint field", Kimi K3), and clients
|
||||
like Claude Code attach cache_control to every request, so a map-known model without
|
||||
the capability must not receive them on system, message, or tool blocks. Unmapped ids
|
||||
(application inference profile ARNs, models newer than the map) keep emitting so
|
||||
existing caching setups never silently degrade."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
|
|
@ -5787,14 +5792,25 @@ def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model,
|
|||
{"role": "system", "content": [{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}}]},
|
||||
{"role": "user", "content": [{"type": "text", "text": "hi", "cache_control": {"type": "ephemeral"}}]},
|
||||
],
|
||||
optional_params={},
|
||||
optional_params={
|
||||
"tools": [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {"name": "get_weather", "parameters": {"type": "object", "properties": {}}},
|
||||
"cache_control": {"type": "ephemeral"},
|
||||
}
|
||||
]
|
||||
},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert ("cachePoint" in json.dumps(body)) is expects_cache_points
|
||||
assert ("cachePoint" in json.dumps(body["system"])) is expects_cache_points
|
||||
assert ("cachePoint" in json.dumps(body["messages"])) is expects_cache_points
|
||||
assert ("cachePoint" in json.dumps(body["toolConfig"])) is expects_cache_points
|
||||
assert body["system"][0]["text"] == "sys"
|
||||
assert body["messages"][0]["content"][0]["text"] == "hi"
|
||||
assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather"
|
||||
|
||||
|
||||
def test_tool_config_cachepoint_not_placed_or_credited_for_model_without_prompt_caching(monkeypatch):
|
||||
|
|
|
|||
|
|
@ -479,6 +479,75 @@ def test_capability_lookups_fall_back_to_base_model_when_regional_entry_lacks_fi
|
|||
assert bedrock_converse_supports_parallel_tool_use_config(regional) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("entry", "expected"),
|
||||
[
|
||||
pytest.param(
|
||||
{"supports_prompt_caching": True, "supports_prompt_cache_breakpoint": False},
|
||||
False,
|
||||
id="priced-cached-tokens-but-rejects-the-explicit-marker",
|
||||
),
|
||||
pytest.param(
|
||||
{"supports_prompt_caching": False, "supports_prompt_cache_breakpoint": True},
|
||||
True,
|
||||
id="explicit-marker-flag-wins-over-the-caching-flag",
|
||||
),
|
||||
pytest.param({"supports_prompt_caching": True}, True, id="caching-flag-alone-keeps-emitting"),
|
||||
pytest.param({"supports_prompt_caching": False}, False, id="no-caching-and-no-marker-flag"),
|
||||
],
|
||||
)
|
||||
def test_bedrock_model_accepts_cache_points_prefers_the_explicit_breakpoint_flag(monkeypatch, entry, expected):
|
||||
import litellm
|
||||
from litellm.llms.bedrock.common_utils import bedrock_model_accepts_cache_points
|
||||
|
||||
base = "vendor.breakpoint-flag-test"
|
||||
monkeypatch.setitem(litellm.model_cost, f"us.{base}", {"input_cost_per_token": 1e-06})
|
||||
monkeypatch.setitem(litellm.model_cost, base, entry)
|
||||
|
||||
assert bedrock_model_accepts_cache_points(f"us.{base}") is expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["moonshotai.kimi-k3", "us.moonshotai.kimi-k3", "global.moonshotai.kimi-k3"])
|
||||
def test_kimi_k3_keeps_cached_token_pricing_while_refusing_converse_cache_points(model, local_model_cost_map):
|
||||
import litellm
|
||||
from litellm.llms.bedrock.common_utils import bedrock_model_accepts_cache_points
|
||||
|
||||
assert bedrock_model_accepts_cache_points(model) is False
|
||||
assert litellm.utils.supports_prompt_caching(model=model, custom_llm_provider="bedrock") is True
|
||||
assert litellm.model_cost[model]["cache_read_input_token_cost"] > 0
|
||||
|
||||
|
||||
def test_deployment_model_info_breakpoint_flag_covers_an_unmapped_arn(local_model_cost_map):
|
||||
from litellm import Router
|
||||
from litellm.llms.bedrock.common_utils import bedrock_model_accepts_cache_points
|
||||
|
||||
flagged_arn = "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/flagged"
|
||||
unflagged_arn = "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/unflagged"
|
||||
converse_arn = "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/converse"
|
||||
Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "kimi-k3-profile-converse",
|
||||
"litellm_params": {"model": f"bedrock/converse/{converse_arn}", "aws_region_name": "us-east-1"},
|
||||
"model_info": {"supports_prompt_cache_breakpoint": False},
|
||||
},
|
||||
{
|
||||
"model_name": "kimi-k3-profile",
|
||||
"litellm_params": {"model": f"bedrock/{flagged_arn}", "aws_region_name": "us-east-1"},
|
||||
"model_info": {"supports_prompt_cache_breakpoint": False},
|
||||
},
|
||||
{
|
||||
"model_name": "kimi-k3-profile-unflagged",
|
||||
"litellm_params": {"model": f"bedrock/{unflagged_arn}", "aws_region_name": "us-east-1"},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
assert bedrock_model_accepts_cache_points(flagged_arn) is False
|
||||
assert bedrock_model_accepts_cache_points(converse_arn) is False
|
||||
assert bedrock_model_accepts_cache_points(unflagged_arn) is True
|
||||
|
||||
|
||||
def test_merge_bedrock_aws_request_params_strips_caller_identity_when_deployment_has_static_credentials():
|
||||
from litellm.llms.bedrock.common_utils import merge_bedrock_aws_request_params
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue