fix(bedrock): stop emitting Converse cachePoint blocks for Kimi K3 (#44292)

* fix(bedrock): stop emitting Converse cachePoint blocks for Kimi K3

Bedrock prices Kimi K3 cache reads through implicit caching but Converse
rejects the explicit cachePoint marker ("This model doesn't support the
cachePoint field"), so any cache_control on the request answered 400.
Mark the three K3 rows supports_prompt_cache_breakpoint: false and have
bedrock_model_accepts_cache_points honor that flag before falling back to
supports_prompt_caching, keeping cached-token pricing intact.

* test(bedrock): assert cache points per request section

* fix(bedrock): honor a deployment's cache breakpoint flag for unmapped models

* fix(bedrock): read a converse-routed deployment's cache breakpoint flag

* refactor(bedrock): look up cache breakpoint flags by key

* test(bedrock): add the Kimi K3 cache point wire audit

---------

Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-03 18:11:46 +00:00 • committed by GitHub
parent 4732647de2
commit 6c32384d8c
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 1699 additions and 17 deletions

View file

@ -1125,23 +1125,42 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
models without prompt caching support ("You invoked an unsupported model or your
request did not allow prompt caching"), so a model whose cost-map entry does not declare
``supports_prompt_caching`` must not receive them. A model absent from the map
(an application inference profile ARN, a model newer than the map) keeps emitting
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
is not reusable here: it returns False for unmapped models, the opposite polarity.
``supports_prompt_caching`` must not receive them. An explicit
``supports_prompt_cache_breakpoint`` on the entry wins over that flag: a model can price
cached tokens through implicit caching yet reject the marker on Converse ("This model
doesn't support the cachePoint field", Kimi K3). The router registers a deployment's
``model_info`` under ``bedrock/<model>`` as configured, route prefix included, while the
Converse transformation sees the model with ``converse/`` or ``converse_like/`` already
stripped, so every registration form is read. That flag set there covers an application
inference profile ARN or a model newer than the map, while only the map decides whether
a model is known: absent a map entry the model keeps emitting so existing caching setups
never silently degrade. ``litellm.utils.supports_prompt_caching`` is not reusable here:
it returns False for unmapped models, the opposite polarity.
"""
if model is None:
return True
if _OPENAI_FAMILY_MODEL_RE.search(model):
return False
entries: Final = tuple(
entry
for candidate in (model, get_bedrock_base_model(model))
if (entry := litellm.model_cost.get(candidate)) is not None
map_keys: Final = (model, get_bedrock_base_model(model))
registered_keys: Final = tuple(f"bedrock/{route}{model}" for route in ("", "converse/", "converse_like/"))
explicit_marker_support: Final = next(
(
entry.get("supports_prompt_cache_breakpoint") is True
for key in (*registered_keys, *map_keys)
if (entry := litellm.model_cost.get(key)) is not None
and entry.get("supports_prompt_cache_breakpoint") is not None
),
None,
)
if not entries:
if explicit_marker_support is not None:
return explicit_marker_support
if not any(key in litellm.model_cost for key in map_keys):
return True
return any(entry.get("supports_prompt_caching") is True for entry in entries)
return any(
entry.get("supports_prompt_caching") is True
for key in map_keys
if (entry := litellm.model_cost.get(key)) is not None
)
def bedrock_supports_tool_search(model: str) -> bool:

View file

@ -76931,6 +76931,7 @@
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json",
"supports_audio_input": false,
"supports_function_calling": true,
"supports_prompt_cache_breakpoint": false,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -76952,6 +76953,7 @@
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_function_calling": true,
"supports_prompt_cache_breakpoint": false,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -76973,6 +76975,7 @@
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_function_calling": true,
"supports_prompt_cache_breakpoint": false,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,

View file

@ -76931,6 +76931,7 @@
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json",
"supports_audio_input": false,
"supports_function_calling": true,
"supports_prompt_cache_breakpoint": false,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -76952,6 +76953,7 @@
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_function_calling": true,
"supports_prompt_cache_breakpoint": false,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -76973,6 +76975,7 @@
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_audio_input": false,
"supports_function_calling": true,
"supports_prompt_cache_breakpoint": false,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,

File diff suppressed because it is too large Load diff

View file

@ -5769,15 +5769,20 @@ def test_cache_control_injection_tool_config_drops_ttl_for_unsupported_model():
pytest.param("global.openai.gpt-6-astra", False, id="openai-family-implicit-caching-only"),
pytest.param("openai.gpt-oss-120b-1:0", False, id="openai-gpt-oss"),
pytest.param("us.openai.gpt-99-unmapped", False, id="unmapped-openai-family-still-suppressed"),
pytest.param("us.moonshotai.kimi-k3", False, id="kimi-k3-prices-cached-tokens-but-rejects-cachepoint"),
pytest.param("global.moonshotai.kimi-k3", False, id="kimi-k3-global-profile"),
pytest.param("us-east-1/us.moonshotai.kimi-k3", False, id="kimi-k3-regional-route-resolves-through-profile"),
],
)
def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model, expects_cache_points, monkeypatch):
"""Bedrock rejects cachePoint blocks for models without prompt caching support
("You invoked an unsupported model or your request did not allow prompt caching"),
and clients like Claude Code attach cache_control to every request, so a map-known
model without the capability must not receive them. Unmapped ids (application
inference profile ARNs, models newer than the map) keep emitting so existing
caching setups never silently degrade."""
("You invoked an unsupported model or your request did not allow prompt caching")
and for models that price cached tokens yet take the marker only on their native
endpoints ("This model doesn't support the cachePoint field", Kimi K3), and clients
like Claude Code attach cache_control to every request, so a map-known model without
the capability must not receive them on system, message, or tool blocks. Unmapped ids
(application inference profile ARNs, models newer than the map) keep emitting so
existing caching setups never silently degrade."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
@ -5787,14 +5792,25 @@ def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model,
{"role": "system", "content": [{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}}]},
{"role": "user", "content": [{"type": "text", "text": "hi", "cache_control": {"type": "ephemeral"}}]},
],
optional_params={},
optional_params={
"tools": [
{
"type": "function",
"function": {"name": "get_weather", "parameters": {"type": "object", "properties": {}}},
"cache_control": {"type": "ephemeral"},
}
]
},
litellm_params={},
headers={},
)
assert ("cachePoint" in json.dumps(body)) is expects_cache_points
assert ("cachePoint" in json.dumps(body["system"])) is expects_cache_points
assert ("cachePoint" in json.dumps(body["messages"])) is expects_cache_points
assert ("cachePoint" in json.dumps(body["toolConfig"])) is expects_cache_points
assert body["system"][0]["text"] == "sys"
assert body["messages"][0]["content"][0]["text"] == "hi"
assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather"
def test_tool_config_cachepoint_not_placed_or_credited_for_model_without_prompt_caching(monkeypatch):

View file

@ -479,6 +479,75 @@ def test_capability_lookups_fall_back_to_base_model_when_regional_entry_lacks_fi
assert bedrock_converse_supports_parallel_tool_use_config(regional) is True
@pytest.mark.parametrize(
("entry", "expected"),
[
pytest.param(
{"supports_prompt_caching": True, "supports_prompt_cache_breakpoint": False},
False,
id="priced-cached-tokens-but-rejects-the-explicit-marker",
),
pytest.param(
{"supports_prompt_caching": False, "supports_prompt_cache_breakpoint": True},
True,
id="explicit-marker-flag-wins-over-the-caching-flag",
),
pytest.param({"supports_prompt_caching": True}, True, id="caching-flag-alone-keeps-emitting"),
pytest.param({"supports_prompt_caching": False}, False, id="no-caching-and-no-marker-flag"),
],
)
def test_bedrock_model_accepts_cache_points_prefers_the_explicit_breakpoint_flag(monkeypatch, entry, expected):
import litellm
from litellm.llms.bedrock.common_utils import bedrock_model_accepts_cache_points
base = "vendor.breakpoint-flag-test"
monkeypatch.setitem(litellm.model_cost, f"us.{base}", {"input_cost_per_token": 1e-06})
monkeypatch.setitem(litellm.model_cost, base, entry)
assert bedrock_model_accepts_cache_points(f"us.{base}") is expected
@pytest.mark.parametrize("model", ["moonshotai.kimi-k3", "us.moonshotai.kimi-k3", "global.moonshotai.kimi-k3"])
def test_kimi_k3_keeps_cached_token_pricing_while_refusing_converse_cache_points(model, local_model_cost_map):
import litellm
from litellm.llms.bedrock.common_utils import bedrock_model_accepts_cache_points
assert bedrock_model_accepts_cache_points(model) is False
assert litellm.utils.supports_prompt_caching(model=model, custom_llm_provider="bedrock") is True
assert litellm.model_cost[model]["cache_read_input_token_cost"] > 0
def test_deployment_model_info_breakpoint_flag_covers_an_unmapped_arn(local_model_cost_map):
from litellm import Router
from litellm.llms.bedrock.common_utils import bedrock_model_accepts_cache_points
flagged_arn = "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/flagged"
unflagged_arn = "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/unflagged"
converse_arn = "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/converse"
Router(
model_list=[
{
"model_name": "kimi-k3-profile-converse",
"litellm_params": {"model": f"bedrock/converse/{converse_arn}", "aws_region_name": "us-east-1"},
"model_info": {"supports_prompt_cache_breakpoint": False},
},
{
"model_name": "kimi-k3-profile",
"litellm_params": {"model": f"bedrock/{flagged_arn}", "aws_region_name": "us-east-1"},
"model_info": {"supports_prompt_cache_breakpoint": False},
},
{
"model_name": "kimi-k3-profile-unflagged",
"litellm_params": {"model": f"bedrock/{unflagged_arn}", "aws_region_name": "us-east-1"},
},
]
)
assert bedrock_model_accepts_cache_points(flagged_arn) is False
assert bedrock_model_accepts_cache_points(converse_arn) is False
assert bedrock_model_accepts_cache_points(unflagged_arn) is True
def test_merge_bedrock_aws_request_params_strips_caller_identity_when_deployment_has_static_credentials():
from litellm.llms.bedrock.common_utils import merge_bedrock_aws_request_params