fix: don't retire a completed batch from cost recovery while output_file_id is still lagging

This commit is contained in:
72004 2026-08-21 02:05:05 +05:00
parent fc3b160fb5
commit f8b31f493a
2 changed files with 62 additions and 1 deletions

View file

@ -1288,6 +1288,25 @@ def batch_cost_poller_is_active() -> bool:
return False
def _completed_batch_safe_to_retire(response) -> bool:
"""Whether a "completed" batch may be retired from cost recovery.
``batch_processed=True`` is the sole re-pickup gate for CheckBatchCost's
cost-recovery poller, so setting it retires the batch permanently. A batch can
reach ``status="completed"`` while ``output_file_id`` is still ``None`` (the
provider response briefly lags before the output id populates). Retiring in that
window loses the spend record forever. Retire only once we can prove there is
nothing left to recover: the output file has actually arrived, or the provider
reports no successful request lines. When counts are unknown, stay eligible so
the next poller pass revisits it. (#37713)
"""
if getattr(response, "output_file_id", None) is not None:
return True
request_counts = getattr(response, "request_counts", None)
completed = getattr(request_counts, "completed", None)
return completed == 0
async def update_batch_in_database(
batch_id: str,
unified_batch_id: str | Literal[False],
@ -1369,7 +1388,7 @@ async def update_batch_in_database(
}
poller_owns: Final = batch_cost_poller_is_active() if poller_owns_accounting is None else poller_owns_accounting
if db_status == "complete" and not poller_owns:
if db_status == "complete" and not poller_owns and _completed_batch_safe_to_retire(response):
update_data["batch_processed"] = True
try:

View file

@ -431,3 +431,45 @@ def test_add_internal_model_credentials_survives_a_failing_deployment_lookup():
add_internal_model_credentials(data=data, llm_router=router, model_id="deployment-gone")
assert data == {"batch_id": "unified-batch-id"}
from litellm.proxy.openai_files_endpoints.common_utils import (
_completed_batch_safe_to_retire,
)
def _completed_batch(output_file_id, completed=None) -> LiteLLMBatch:
kwargs = dict(
id="batch-1",
completion_window="24h",
created_at=1234567890,
endpoint="/v1/chat/completions",
input_file_id="file-in",
object="batch",
status="completed",
output_file_id=output_file_id,
error_file_id=None,
)
if completed is not None:
kwargs["request_counts"] = {"total": completed, "completed": completed, "failed": 0}
return LiteLLMBatch(**kwargs)
class TestCompletedBatchSafeToRetire:
"""A completed batch is only safe to retire from cost recovery once its output
file has arrived or the provider proves no successful lines (#37713)."""
def test_output_file_present_is_safe(self):
assert _completed_batch_safe_to_retire(_completed_batch("file-out")) is True
def test_no_output_and_no_successful_lines_is_safe(self):
# Every request line errored -> nothing left to recover.
assert _completed_batch_safe_to_retire(_completed_batch(None, completed=0)) is True
def test_no_output_but_successful_lines_is_not_safe(self):
# The bug: output_file_id is lagging; retiring here loses the spend record.
assert _completed_batch_safe_to_retire(_completed_batch(None, completed=5)) is False
def test_no_output_and_unknown_counts_is_not_safe(self):
# Counts unknown -> stay eligible so the next poller pass revisits it.
assert _completed_batch_safe_to_retire(_completed_batch(None)) is False