fix(batches): only emit line items for final batches

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
shivam 2026-09-17 22:50:34 +00:00
parent b38b867918
commit 0ea8ce740c
4 changed files with 46 additions and 8 deletions

View file

@ -3027,7 +3027,7 @@ class Logging(LiteLLMLoggingBaseClass):
cost_for_built_in_tools_cost_usd_dollar=0.0,
)
if litellm.store_batch_line_items_in_callbacks:
if litellm.store_batch_line_items_in_callbacks and (has_explicit_batch_data or should_compute_batch_data):
from litellm.batches.batch_line_item_logging import log_batch_line_items
await log_batch_line_items(

View file

@ -34,6 +34,7 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
def error_response_text(response: httpx.Response) -> str:
@ -878,9 +879,10 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
Whether Converse ``cachePoint`` blocks may be sent to this model.
Bedrock rejects requests carrying cachePoint blocks for models without prompt
caching support ("You invoked an unsupported model or your request did not allow
prompt caching"), so a model whose cost-map entry does not declare
OpenAI-family models only support implicit caching and never accept explicit
``cachePoint`` blocks. Bedrock rejects requests carrying cachePoint blocks for
models without prompt caching support ("You invoked an unsupported model or your
request did not allow prompt caching"), so a model whose cost-map entry does not declare
``supports_prompt_caching`` must not receive them. A model absent from the map
(an application inference profile ARN, a model newer than the map) keeps emitting
so existing caching setups never silently degrade. ``litellm.utils.supports_prompt_caching``
@ -888,6 +890,8 @@ def bedrock_model_accepts_cache_points(model: str | None) -> bool:
"""
if model is None:
return True
if _OPENAI_FAMILY_MODEL_RE.search(model):
return False
entries: Final = tuple(
entry
for candidate in (model, get_bedrock_base_model(model))

View file

@ -15,6 +15,7 @@ import json
import uuid
from datetime import datetime
from types import SimpleNamespace
from typing import Final
from unittest.mock import AsyncMock, patch
import pytest
@ -222,6 +223,29 @@ async def test_flag_off_emits_only_aggregate(recorder):
file_mock.assert_not_called()
@pytest.mark.asyncio
async def test_in_progress_batch_poll_emits_no_line_events(recorder):
litellm.store_batch_line_items_in_callbacks = True # test-quality-ok: the flag under test is a module global; fixture restores it
in_progress: Final = LiteLLMBatch(
id="batch_wip",
object="batch",
endpoint="/v1/chat/completions",
input_file_id="input-file-1",
output_file_id=None,
error_file_id=None,
status="in_progress",
completion_window="24h",
created_at=1,
)
file_mock: Final = AsyncMock(side_effect=_file_content)
with patch("litellm.files.main.afile_content", file_mock): # test-quality-ok: afile_content is the provider boundary; no injection seam for managed file fetch
await _parent_logging().async_success_handler(result=in_progress)
file_mock.assert_not_called()
assert all(_hidden(e).get("batch_custom_id") is None for e in recorder.success_events)
assert len(recorder.failure_events) == 0
@pytest.mark.asyncio
async def test_input_fetch_failure_still_emits_aggregate(recorder):
litellm.store_batch_line_items_in_callbacks = True # test-quality-ok: the flag under test is a module global; fixture restores it

View file

@ -1189,17 +1189,24 @@ def test_get_supported_openai_params_bedrock_converse():
@pytest.mark.parametrize(
"tools, expected_marker",
"tools, model, expected_marker",
[
pytest.param(
[{"type": "function", "function": {"name": "f", "parameters": {"type": "object", "properties": {}}}}],
"anthropic.claude-sonnet-4-5-20250929-v1:0",
"dep-bedrock",
id="tools-present-so-the-cachepoint-is-placed",
),
pytest.param(None, None, id="no-tools-so-nothing-is-placed"),
pytest.param(None, "anthropic.claude-sonnet-4-5-20250929-v1:0", None, id="no-tools-so-nothing-is-placed"),
pytest.param(
[{"type": "function", "function": {"name": "f", "parameters": {"type": "object", "properties": {}}}}],
"global.openai.gpt-6-astra",
None,
id="openai-family-implicit-caching-only",
),
],
)
def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, expected_marker):
def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, model, expected_marker):
"""Spend attribution credits the gateway for breakpoints it placed, and a tool_config
point becomes one here or nowhere.
@ -1213,7 +1220,7 @@ def test_tool_config_cachepoint_is_credited_only_where_it_is_placed(tools, expec
optional_params["tools"] = tools
data = AmazonConverseConfig()._transform_request_helper(
model="anthropic.claude-sonnet-4-5-20250929-v1:0",
model=model,
system_content_blocks=[],
optional_params=optional_params,
messages=[{"role": "user", "content": "hi"}],
@ -5591,6 +5598,9 @@ def test_cache_control_injection_tool_config_drops_ttl_for_unsupported_model():
True,
id="unmapped-arn-keeps-emitting",
),
pytest.param("global.openai.gpt-6-astra", False, id="openai-family-implicit-caching-only"),
pytest.param("openai.gpt-oss-120b-1:0", False, id="openai-gpt-oss"),
pytest.param("us.openai.gpt-99-unmapped", False, id="unmapped-openai-family-still-suppressed"),
],
)
def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model, expects_cache_points, monkeypatch):