fix(bedrock): never emit Converse cachePoint for OpenAI-family models

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Devin AI 2026-09-16 13:21:17 +00:00
parent a8979fe054
commit 1674a3d767
2 changed files with 21 additions and 7 deletions

View file

@ -34,6 +34,7 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
def error_response_text(response: httpx.Response) -> str:
@ -878,9 +879,10 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
Whether Converse ``cachePoint`` blocks may be sent to this model.
Bedrock rejects requests carrying cachePoint blocks for models without prompt
caching support ("You invoked an unsupported model or your request did not allow
prompt caching"), so a model whose cost-map entry does not declare
OpenAI-family models only support implicit caching and never accept explicit
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
models without prompt caching support ("You invoked an unsupported model or your
request did not allow prompt caching"), so a model whose cost-map entry does not declare
``supports_prompt_caching`` must not receive them. A model absent from the map
(an application inference profile ARN, a model newer than the map) keeps emitting
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
@ -888,6 +890,8 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
if model is None:
return True
if _OPENAI_FAMILY_MODEL_RE.search(model):
return False
entries: Final = tuple(
entry
for candidate in (model, get_bedrock_base_model(model))

View file

@ -1077,17 +1077,24 @@ def test_get_supported_openai_params_bedrock_converse():
@pytest.mark.parametrize(
"tools, expected_marker",
"tools, model, expected_marker",
[
pytest.param(
[{"type": "function", "function": {"name": "f", "parameters": {"type": "object", "properties": {}}}}],
"anthropic.claude-sonnet-4-5-20250929-v1:0",
"dep-bedrock",
id="tools-present-so-the-cachepoint-is-placed",
),
pytest.param(None, None, id="no-tools-so-nothing-is-placed"),
pytest.param(None, "anthropic.claude-sonnet-4-5-20250929-v1:0", None, id="no-tools-so-nothing-is-placed"),
pytest.param(
[{"type": "function", "function": {"name": "f", "parameters": {"type": "object", "properties": {}}}}],
"global.openai.gpt-6-astra",
None,
id="openai-family-implicit-caching-only",
),
],
)
def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, expected_marker):
def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, model, expected_marker):
"""Spend attribution credits the gateway for breakpoints it placed, and a tool_config
point becomes one here or nowhere.
@ -1101,7 +1108,7 @@ def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, expec
optional_params["tools"] = tools
data = AmazonConverseConfig()._transform_request_helper(
model="anthropic.claude-sonnet-4-5-20250929-v1:0",
model=model,
system_content_blocks=[],
optional_params=optional_params,
messages=[{"role": "user", "content": "hi"}],
@ -5479,6 +5486,9 @@ def test_cache_control_injection_tool_config_drops_ttl_for_unsupported_model():
True,
id="unmapped-arn-keeps-emitting",
),
pytest.param("global.openai.gpt-6-astra", False, id="openai-family-implicit-caching-only"),
pytest.param("openai.gpt-oss-120b-1:0", False, id="openai-gpt-oss"),
pytest.param("us.openai.gpt-99-unmapped", False, id="unmapped-openai-family-still-suppressed"),
],
)
def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model, expects_cache_points, monkeypatch):