mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-12 23:01:41 +00:00
fix: don't retire a completed batch from cost recovery while output_file_id is still lagging
This commit is contained in:
parent
fc3b160fb5
commit
f8b31f493a
2 changed files with 62 additions and 1 deletions
|
|
@ -1288,6 +1288,25 @@ def batch_cost_poller_is_active() -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def _completed_batch_safe_to_retire(response) -> bool:
|
||||
"""Whether a "completed" batch may be retired from cost recovery.
|
||||
|
||||
``batch_processed=True`` is the sole re-pickup gate for CheckBatchCost's
|
||||
cost-recovery poller, so setting it retires the batch permanently. A batch can
|
||||
reach ``status="completed"`` while ``output_file_id`` is still ``None`` (the
|
||||
provider response briefly lags before the output id populates). Retiring in that
|
||||
window loses the spend record forever. Retire only once we can prove there is
|
||||
nothing left to recover: the output file has actually arrived, or the provider
|
||||
reports no successful request lines. When counts are unknown, stay eligible so
|
||||
the next poller pass revisits it. (#37713)
|
||||
"""
|
||||
if getattr(response, "output_file_id", None) is not None:
|
||||
return True
|
||||
request_counts = getattr(response, "request_counts", None)
|
||||
completed = getattr(request_counts, "completed", None)
|
||||
return completed == 0
|
||||
|
||||
|
||||
async def update_batch_in_database(
|
||||
batch_id: str,
|
||||
unified_batch_id: str | Literal[False],
|
||||
|
|
@ -1369,7 +1388,7 @@ async def update_batch_in_database(
|
|||
}
|
||||
|
||||
poller_owns: Final = batch_cost_poller_is_active() if poller_owns_accounting is None else poller_owns_accounting
|
||||
if db_status == "complete" and not poller_owns:
|
||||
if db_status == "complete" and not poller_owns and _completed_batch_safe_to_retire(response):
|
||||
update_data["batch_processed"] = True
|
||||
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -431,3 +431,45 @@ def test_add_internal_model_credentials_survives_a_failing_deployment_lookup():
|
|||
add_internal_model_credentials(data=data, llm_router=router, model_id="deployment-gone")
|
||||
|
||||
assert data == {"batch_id": "unified-batch-id"}
|
||||
|
||||
|
||||
from litellm.proxy.openai_files_endpoints.common_utils import (
|
||||
_completed_batch_safe_to_retire,
|
||||
)
|
||||
|
||||
|
||||
def _completed_batch(output_file_id, completed=None) -> LiteLLMBatch:
|
||||
kwargs = dict(
|
||||
id="batch-1",
|
||||
completion_window="24h",
|
||||
created_at=1234567890,
|
||||
endpoint="/v1/chat/completions",
|
||||
input_file_id="file-in",
|
||||
object="batch",
|
||||
status="completed",
|
||||
output_file_id=output_file_id,
|
||||
error_file_id=None,
|
||||
)
|
||||
if completed is not None:
|
||||
kwargs["request_counts"] = {"total": completed, "completed": completed, "failed": 0}
|
||||
return LiteLLMBatch(**kwargs)
|
||||
|
||||
|
||||
class TestCompletedBatchSafeToRetire:
|
||||
"""A completed batch is only safe to retire from cost recovery once its output
|
||||
file has arrived or the provider proves no successful lines (#37713)."""
|
||||
|
||||
def test_output_file_present_is_safe(self):
|
||||
assert _completed_batch_safe_to_retire(_completed_batch("file-out")) is True
|
||||
|
||||
def test_no_output_and_no_successful_lines_is_safe(self):
|
||||
# Every request line errored -> nothing left to recover.
|
||||
assert _completed_batch_safe_to_retire(_completed_batch(None, completed=0)) is True
|
||||
|
||||
def test_no_output_but_successful_lines_is_not_safe(self):
|
||||
# The bug: output_file_id is lagging; retiring here loses the spend record.
|
||||
assert _completed_batch_safe_to_retire(_completed_batch(None, completed=5)) is False
|
||||
|
||||
def test_no_output_and_unknown_counts_is_not_safe(self):
|
||||
# Counts unknown -> stay eligible so the next poller pass revisits it.
|
||||
assert _completed_batch_safe_to_retire(_completed_batch(None)) is False
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue