mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(interactions): release the reservation for creates nothing will poll, and let OTEL see the settled cost
Two review findings, both in the handoff between the create's success callback and the background poll task. The callback deferred its budget reservation release for any interactions response with no usage, but the scheduler only starts a poll task when the status is in_progress and an id is present. A create that came back terminal without usage therefore matched the callback's test, got no poll task, and left its reservation open forever: the pre-call estimate stayed added to the key, user, team and org spend counters, and the key began refusing traffic against budget it had never spent. The two conditions now come from one shared gate so they cannot drift apart again. Settling a background interaction re-runs the success handlers for a second result on the same request, and OTEL dedupes span emission on a marker held in that request's metadata. The in-progress create claimed the marker, so the completion, the only event carrying usage and cost, was dropped as a duplicate by OTEL and by every integration deriving from it. Clearing the success-scoped markers alongside the existing dedup flag lets the cost span through, leaving failure and guardrail markers untouched.
This commit is contained in:
parent
9906770e41
commit
94caab7302
5 changed files with 180 additions and 4 deletions
|
|
@ -162,6 +162,17 @@ async def _release_open_budget_reservation(logging_obj: "LiteLLMLoggingObj") ->
|
|||
verbose_logger.exception("Failed to release budget reservation for an unbilled background interaction")
|
||||
|
||||
|
||||
def is_pollable_background_interaction(response: InteractionsAPIResponse) -> bool:
|
||||
"""
|
||||
The single gate deciding whether a create's response gets a poll task.
|
||||
The proxy's success callback defers releasing the budget reservation for
|
||||
exactly these responses, on the promise that a poll task will settle them,
|
||||
so a response one site accepts and the other refuses strands its
|
||||
reservation on the spend counters with nothing left to reconcile it.
|
||||
"""
|
||||
return response.status == "in_progress" and bool(response.id)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _ActiveBackgroundPoll:
|
||||
task: "asyncio.Task[None]"
|
||||
|
|
@ -188,7 +199,7 @@ def maybe_schedule_background_interaction_cost_polling(
|
|||
return None
|
||||
if not isinstance(response, InteractionsAPIResponse):
|
||||
return None
|
||||
if response.status != "in_progress" or not response.id:
|
||||
if not is_pollable_background_interaction(response):
|
||||
return None
|
||||
logging_obj = create_kwargs.get("litellm_logging_obj")
|
||||
if not isinstance(logging_obj, Logging):
|
||||
|
|
|
|||
|
|
@ -2200,12 +2200,37 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
Log the terminal result of a background interaction as a fresh success
|
||||
event. The create request already ran success logging for its
|
||||
``in_progress`` response (no usage, so no cost was tracked); clearing
|
||||
the dedup flag lets the completed result flow through cost calculation
|
||||
the dedup flags lets the completed result flow through cost calculation
|
||||
and spend tracking exactly once, spanning create to completion.
|
||||
"""
|
||||
self.model_call_details.pop("has_logged_async_success", None)
|
||||
self._reset_success_emission_dedupe()
|
||||
await self.async_success_handler(result=result)
|
||||
|
||||
def _reset_success_emission_dedupe(self) -> None:
|
||||
"""
|
||||
Success callbacks dedupe per request, because the sync and async
|
||||
handlers both fire on some paths and would otherwise report one call
|
||||
twice. A settled background interaction is a genuinely second success
|
||||
event on the same request, so every such marker has to be cleared or
|
||||
the completion, the only event that carries usage and cost, is
|
||||
discarded as a duplicate of the in-progress create.
|
||||
"""
|
||||
self.model_call_details.pop("has_logged_async_success", None)
|
||||
litellm_params = self.model_call_details.get("litellm_params")
|
||||
if not isinstance(litellm_params, dict):
|
||||
return
|
||||
metadata = litellm_params.get("metadata")
|
||||
if not isinstance(metadata, dict):
|
||||
return
|
||||
otel_internal = metadata.get("_otel_internal")
|
||||
if not isinstance(otel_internal, dict):
|
||||
return
|
||||
spans_logged = otel_internal.get("spans_logged")
|
||||
if not isinstance(spans_logged, dict):
|
||||
return
|
||||
for scope in [key for key in spans_logged if isinstance(key, tuple) and key[-1:] == ("success",)]:
|
||||
del spans_logged[scope]
|
||||
|
||||
def _flush_passthrough_collected_chunks_helper(
|
||||
self,
|
||||
raw_bytes: list[bytes],
|
||||
|
|
|
|||
|
|
@ -478,9 +478,12 @@ def _write_spend_metadata_to_kwargs(kwargs: dict, metadata: dict) -> None:
|
|||
|
||||
|
||||
def _is_unbilled_in_progress_interaction(completion_response: object) -> bool:
|
||||
from litellm.interactions.background_cost_polling import is_pollable_background_interaction
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
return isinstance(completion_response, InteractionsAPIResponse) and completion_response.usage is None
|
||||
if not isinstance(completion_response, InteractionsAPIResponse):
|
||||
return False
|
||||
return completion_response.usage is None and is_pollable_background_interaction(completion_response)
|
||||
|
||||
|
||||
def _should_track_cost_callback(
|
||||
|
|
|
|||
|
|
@ -4644,6 +4644,44 @@ async def test_background_interaction_completion_rebills_after_in_progress_succe
|
|||
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_background_interaction_completion_lets_otel_emit_the_cost_span():
|
||||
"""
|
||||
OTEL, and every integration that derives from it, dedupes span emission on
|
||||
a marker kept in the request's own metadata. The in-progress create claims
|
||||
that marker, so without clearing it the settled completion, the only event
|
||||
carrying usage and cost, is discarded as a duplicate and every
|
||||
OTEL-family backend shows the interaction as a span with no cost at all.
|
||||
"""
|
||||
import datetime as dt
|
||||
|
||||
from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
otel = OpenTelemetry(config=OpenTelemetryConfig(exporter="console"))
|
||||
logging_obj = _interactions_logging_obj(stream=False)
|
||||
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
|
||||
await logging_obj.async_success_handler(
|
||||
result=in_progress,
|
||||
start_time=dt.datetime.now(),
|
||||
end_time=dt.datetime.now(),
|
||||
)
|
||||
|
||||
assert otel._emit_once(logging_obj.model_call_details, "success") is True
|
||||
assert otel._emit_once(logging_obj.model_call_details, "success") is False
|
||||
|
||||
completed = InteractionsAPIResponse(
|
||||
id="interactions/abc",
|
||||
model="gemini-2.5-flash",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage=dict(INTERACTIONS_USAGE_BLOCK),
|
||||
)
|
||||
await logging_obj.async_log_background_interaction_completion(result=completed)
|
||||
|
||||
assert otel._emit_once(logging_obj.model_call_details, "success") is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"call_type",
|
||||
["aget", "get", "aget_interaction", "adelete_interaction", "acancel_interaction"],
|
||||
|
|
|
|||
|
|
@ -859,6 +859,105 @@ async def test_track_cost_callback_releases_reservation_for_in_progress_interact
|
|||
mock_proxy_logging.failed_tracking_alert.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"status",
|
||||
["completed", "failed", "cancelled", "incomplete", "requires_action", "budget_exceeded"],
|
||||
)
|
||||
async def test_track_cost_callback_releases_reservation_for_unpollable_interaction(status):
|
||||
"""
|
||||
Only an in-progress create gets a poll task, so a create that comes back
|
||||
terminal with no usage has nobody left to reconcile its reservation. The
|
||||
callback must release it there and then, or the pre-call estimate stays
|
||||
added to the key, user, team and org spend counters and starts refusing
|
||||
traffic against budget that was never actually spent.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
terminal_response = InteractionsAPIResponse(
|
||||
id="interactions/bg-abc",
|
||||
model="gemini-3-flash-preview",
|
||||
status=status,
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=terminal_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_track_cost_callback_releases_reservation_for_interaction_without_an_id():
|
||||
"""
|
||||
The scheduler also refuses a response with no id, since it has nothing to
|
||||
poll for, so the callback must not defer to a poll task that will never
|
||||
exist.
|
||||
"""
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
logger = _ProxyDBLogger()
|
||||
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
|
||||
idless_response = InteractionsAPIResponse(
|
||||
id="",
|
||||
model="gemini-3-flash-preview",
|
||||
status="in_progress",
|
||||
)
|
||||
|
||||
with patch(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
) as mock_proxy_logging:
|
||||
mock_proxy_logging.failed_tracking_alert = AsyncMock()
|
||||
mock_proxy_logging.db_spend_update_writer = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=_in_progress_interaction_kwargs(reservation),
|
||||
completion_response=idless_response,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert reservation["finalized"] is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"status",
|
||||
["in_progress", "completed", "failed", "cancelled", "incomplete", "requires_action"],
|
||||
)
|
||||
@pytest.mark.parametrize("interaction_id", ["interactions/bg-abc", ""])
|
||||
def test_callback_defers_exactly_the_interactions_the_scheduler_polls(status, interaction_id):
|
||||
"""
|
||||
Pins the invariant the two modules share: the callback may only hold a
|
||||
budget reservation open for a response the scheduler will actually poll.
|
||||
Any drift between the two gates leaks reservations onto live spend
|
||||
counters, so assert they agree rather than restating either condition.
|
||||
"""
|
||||
from litellm.interactions.background_cost_polling import is_pollable_background_interaction
|
||||
from litellm.proxy.hooks.proxy_track_cost_callback import _is_unbilled_in_progress_interaction
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
response = InteractionsAPIResponse(
|
||||
id=interaction_id,
|
||||
model="gemini-3-flash-preview",
|
||||
status=status,
|
||||
)
|
||||
|
||||
assert _is_unbilled_in_progress_interaction(response) is is_pollable_background_interaction(response)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_post_call_failure_hook_propagates_trace_id_from_logging_obj():
|
||||
"""
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue