From 94caab7302ba471647bfba51bb1e5f8ad4cc5222 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 22 Aug 2026 11:46:18 -0700 Subject: [PATCH] fix(interactions): release the reservation for creates nothing will poll, and let OTEL see the settled cost Two review findings, both in the handoff between the create's success callback and the background poll task. The callback deferred its budget reservation release for any interactions response with no usage, but the scheduler only starts a poll task when the status is in_progress and an id is present. A create that came back terminal without usage therefore matched the callback's test, got no poll task, and left its reservation open forever: the pre-call estimate stayed added to the key, user, team and org spend counters, and the key began refusing traffic against budget it had never spent. The two conditions now come from one shared gate so they cannot drift apart again. Settling a background interaction re-runs the success handlers for a second result on the same request, and OTEL dedupes span emission on a marker held in that request's metadata. The in-progress create claimed the marker, so the completion, the only event carrying usage and cost, was dropped as a duplicate by OTEL and by every integration deriving from it. Clearing the success-scoped markers alongside the existing dedup flag lets the cost span through, leaving failure and guardrail markers untouched. --- .../interactions/background_cost_polling.py | 13 ++- litellm/litellm_core_utils/litellm_logging.py | 29 +++++- .../proxy/hooks/proxy_track_cost_callback.py | 5 +- .../test_litellm_logging.py | 38 +++++++ .../hooks/test_proxy_track_cost_callback.py | 99 +++++++++++++++++++ 5 files changed, 180 insertions(+), 4 deletions(-) diff --git a/litellm/interactions/background_cost_polling.py b/litellm/interactions/background_cost_polling.py index ccf4c2dd853..262d8e3260d 100644 --- a/litellm/interactions/background_cost_polling.py +++ b/litellm/interactions/background_cost_polling.py @@ -162,6 +162,17 @@ async def _release_open_budget_reservation(logging_obj: "LiteLLMLoggingObj") -> verbose_logger.exception("Failed to release budget reservation for an unbilled background interaction") +def is_pollable_background_interaction(response: InteractionsAPIResponse) -> bool: + """ + The single gate deciding whether a create's response gets a poll task. + The proxy's success callback defers releasing the budget reservation for + exactly these responses, on the promise that a poll task will settle them, + so a response one site accepts and the other refuses strands its + reservation on the spend counters with nothing left to reconcile it. + """ + return response.status == "in_progress" and bool(response.id) + + @dataclass(frozen=True, slots=True) class _ActiveBackgroundPoll: task: "asyncio.Task[None]" @@ -188,7 +199,7 @@ def maybe_schedule_background_interaction_cost_polling( return None if not isinstance(response, InteractionsAPIResponse): return None - if response.status != "in_progress" or not response.id: + if not is_pollable_background_interaction(response): return None logging_obj = create_kwargs.get("litellm_logging_obj") if not isinstance(logging_obj, Logging): diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index e7d7213b47c..275803c5aed 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -2200,12 +2200,37 @@ class Logging(LiteLLMLoggingBaseClass): Log the terminal result of a background interaction as a fresh success event. The create request already ran success logging for its ``in_progress`` response (no usage, so no cost was tracked); clearing - the dedup flag lets the completed result flow through cost calculation + the dedup flags lets the completed result flow through cost calculation and spend tracking exactly once, spanning create to completion. """ - self.model_call_details.pop("has_logged_async_success", None) + self._reset_success_emission_dedupe() await self.async_success_handler(result=result) + def _reset_success_emission_dedupe(self) -> None: + """ + Success callbacks dedupe per request, because the sync and async + handlers both fire on some paths and would otherwise report one call + twice. A settled background interaction is a genuinely second success + event on the same request, so every such marker has to be cleared or + the completion, the only event that carries usage and cost, is + discarded as a duplicate of the in-progress create. + """ + self.model_call_details.pop("has_logged_async_success", None) + litellm_params = self.model_call_details.get("litellm_params") + if not isinstance(litellm_params, dict): + return + metadata = litellm_params.get("metadata") + if not isinstance(metadata, dict): + return + otel_internal = metadata.get("_otel_internal") + if not isinstance(otel_internal, dict): + return + spans_logged = otel_internal.get("spans_logged") + if not isinstance(spans_logged, dict): + return + for scope in [key for key in spans_logged if isinstance(key, tuple) and key[-1:] == ("success",)]: + del spans_logged[scope] + def _flush_passthrough_collected_chunks_helper( self, raw_bytes: list[bytes], diff --git a/litellm/proxy/hooks/proxy_track_cost_callback.py b/litellm/proxy/hooks/proxy_track_cost_callback.py index 55d3c6c6ed5..46cf62ede1c 100644 --- a/litellm/proxy/hooks/proxy_track_cost_callback.py +++ b/litellm/proxy/hooks/proxy_track_cost_callback.py @@ -478,9 +478,12 @@ def _write_spend_metadata_to_kwargs(kwargs: dict, metadata: dict) -> None: def _is_unbilled_in_progress_interaction(completion_response: object) -> bool: + from litellm.interactions.background_cost_polling import is_pollable_background_interaction from litellm.types.interactions import InteractionsAPIResponse - return isinstance(completion_response, InteractionsAPIResponse) and completion_response.usage is None + if not isinstance(completion_response, InteractionsAPIResponse): + return False + return completion_response.usage is None and is_pollable_background_interaction(completion_response) def _should_track_cost_callback( diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 8a682517f64..7c90d31261b 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -4644,6 +4644,44 @@ async def test_background_interaction_completion_rebills_after_in_progress_succe assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175 +@pytest.mark.asyncio +async def test_background_interaction_completion_lets_otel_emit_the_cost_span(): + """ + OTEL, and every integration that derives from it, dedupes span emission on + a marker kept in the request's own metadata. The in-progress create claims + that marker, so without clearing it the settled completion, the only event + carrying usage and cost, is discarded as a duplicate and every + OTEL-family backend shows the interaction as a span with no cost at all. + """ + import datetime as dt + + from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig + from litellm.types.interactions import InteractionsAPIResponse + + otel = OpenTelemetry(config=OpenTelemetryConfig(exporter="console")) + logging_obj = _interactions_logging_obj(stream=False) + in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress") + await logging_obj.async_success_handler( + result=in_progress, + start_time=dt.datetime.now(), + end_time=dt.datetime.now(), + ) + + assert otel._emit_once(logging_obj.model_call_details, "success") is True + assert otel._emit_once(logging_obj.model_call_details, "success") is False + + completed = InteractionsAPIResponse( + id="interactions/abc", + model="gemini-2.5-flash", + status="completed", + steps=[], + usage=dict(INTERACTIONS_USAGE_BLOCK), + ) + await logging_obj.async_log_background_interaction_completion(result=completed) + + assert otel._emit_once(logging_obj.model_call_details, "success") is True + + @pytest.mark.parametrize( "call_type", ["aget", "get", "aget_interaction", "adelete_interaction", "acancel_interaction"], diff --git a/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py b/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py index 642db7e2d5d..93e0cbec596 100644 --- a/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py +++ b/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py @@ -859,6 +859,105 @@ async def test_track_cost_callback_releases_reservation_for_in_progress_interact mock_proxy_logging.failed_tracking_alert.assert_not_called() +@pytest.mark.asyncio +@pytest.mark.parametrize( + "status", + ["completed", "failed", "cancelled", "incomplete", "requires_action", "budget_exceeded"], +) +async def test_track_cost_callback_releases_reservation_for_unpollable_interaction(status): + """ + Only an in-progress create gets a poll task, so a create that comes back + terminal with no usage has nobody left to reconcile its reservation. The + callback must release it there and then, or the pre-call estimate stays + added to the key, user, team and org spend counters and starts refusing + traffic against budget that was never actually spent. + """ + from litellm.types.interactions import InteractionsAPIResponse + + logger = _ProxyDBLogger() + reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False} + terminal_response = InteractionsAPIResponse( + id="interactions/bg-abc", + model="gemini-3-flash-preview", + status=status, + ) + + with patch( + "litellm.proxy.proxy_server.proxy_logging_obj", + ) as mock_proxy_logging: + mock_proxy_logging.failed_tracking_alert = AsyncMock() + mock_proxy_logging.db_spend_update_writer = MagicMock() + mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock() + + await logger._PROXY_track_cost_callback( + kwargs=_in_progress_interaction_kwargs(reservation), + completion_response=terminal_response, + start_time=datetime.now(), + end_time=datetime.now(), + ) + + assert reservation["finalized"] is True + + +@pytest.mark.asyncio +async def test_track_cost_callback_releases_reservation_for_interaction_without_an_id(): + """ + The scheduler also refuses a response with no id, since it has nothing to + poll for, so the callback must not defer to a poll task that will never + exist. + """ + from litellm.types.interactions import InteractionsAPIResponse + + logger = _ProxyDBLogger() + reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False} + idless_response = InteractionsAPIResponse( + id="", + model="gemini-3-flash-preview", + status="in_progress", + ) + + with patch( + "litellm.proxy.proxy_server.proxy_logging_obj", + ) as mock_proxy_logging: + mock_proxy_logging.failed_tracking_alert = AsyncMock() + mock_proxy_logging.db_spend_update_writer = MagicMock() + mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock() + + await logger._PROXY_track_cost_callback( + kwargs=_in_progress_interaction_kwargs(reservation), + completion_response=idless_response, + start_time=datetime.now(), + end_time=datetime.now(), + ) + + assert reservation["finalized"] is True + + +@pytest.mark.parametrize( + "status", + ["in_progress", "completed", "failed", "cancelled", "incomplete", "requires_action"], +) +@pytest.mark.parametrize("interaction_id", ["interactions/bg-abc", ""]) +def test_callback_defers_exactly_the_interactions_the_scheduler_polls(status, interaction_id): + """ + Pins the invariant the two modules share: the callback may only hold a + budget reservation open for a response the scheduler will actually poll. + Any drift between the two gates leaks reservations onto live spend + counters, so assert they agree rather than restating either condition. + """ + from litellm.interactions.background_cost_polling import is_pollable_background_interaction + from litellm.proxy.hooks.proxy_track_cost_callback import _is_unbilled_in_progress_interaction + from litellm.types.interactions import InteractionsAPIResponse + + response = InteractionsAPIResponse( + id=interaction_id, + model="gemini-3-flash-preview", + status=status, + ) + + assert _is_unbilled_in_progress_interaction(response) is is_pollable_background_interaction(response) + + @pytest.mark.asyncio async def test_async_post_call_failure_hook_propagates_trace_id_from_logging_obj(): """