fix(interactions): release the reservation for creates nothing will poll, and let OTEL see the settled cost

Two review findings, both in the handoff between the create's success
callback and the background poll task.

The callback deferred its budget reservation release for any interactions
response with no usage, but the scheduler only starts a poll task when the
status is in_progress and an id is present. A create that came back
terminal without usage therefore matched the callback's test, got no poll
task, and left its reservation open forever: the pre-call estimate stayed
added to the key, user, team and org spend counters, and the key began
refusing traffic against budget it had never spent. The two conditions now
come from one shared gate so they cannot drift apart again.

Settling a background interaction re-runs the success handlers for a second
result on the same request, and OTEL dedupes span emission on a marker held
in that request's metadata. The in-progress create claimed the marker, so
the completion, the only event carrying usage and cost, was dropped as a
duplicate by OTEL and by every integration deriving from it. Clearing the
success-scoped markers alongside the existing dedup flag lets the cost span
through, leaving failure and guardrail markers untouched.
This commit is contained in:
mateo-berri 2026-08-22 11:46:18 -07:00
parent 9906770e41
commit 94caab7302
5 changed files with 180 additions and 4 deletions

View file

@ -162,6 +162,17 @@ async def _release_open_budget_reservation(logging_obj: "LiteLLMLoggingObj") ->
verbose_logger.exception("Failed to release budget reservation for an unbilled background interaction")
def is_pollable_background_interaction(response: InteractionsAPIResponse) -> bool:
"""
The single gate deciding whether a create's response gets a poll task.
The proxy's success callback defers releasing the budget reservation for
exactly these responses, on the promise that a poll task will settle them,
so a response one site accepts and the other refuses strands its
reservation on the spend counters with nothing left to reconcile it.
"""
return response.status == "in_progress" and bool(response.id)
@dataclass(frozen=True, slots=True)
class _ActiveBackgroundPoll:
task: "asyncio.Task[None]"
@ -188,7 +199,7 @@ def maybe_schedule_background_interaction_cost_polling(
return None
if not isinstance(response, InteractionsAPIResponse):
return None
if response.status != "in_progress" or not response.id:
if not is_pollable_background_interaction(response):
return None
logging_obj = create_kwargs.get("litellm_logging_obj")
if not isinstance(logging_obj, Logging):

View file

@ -2200,12 +2200,37 @@ class Logging(LiteLLMLoggingBaseClass):
Log the terminal result of a background interaction as a fresh success
event. The create request already ran success logging for its
``in_progress`` response (no usage, so no cost was tracked); clearing
the dedup flag lets the completed result flow through cost calculation
the dedup flags lets the completed result flow through cost calculation
and spend tracking exactly once, spanning create to completion.
"""
self.model_call_details.pop("has_logged_async_success", None)
self._reset_success_emission_dedupe()
await self.async_success_handler(result=result)
def _reset_success_emission_dedupe(self) -> None:
"""
Success callbacks dedupe per request, because the sync and async
handlers both fire on some paths and would otherwise report one call
twice. A settled background interaction is a genuinely second success
event on the same request, so every such marker has to be cleared or
the completion, the only event that carries usage and cost, is
discarded as a duplicate of the in-progress create.
"""
self.model_call_details.pop("has_logged_async_success", None)
litellm_params = self.model_call_details.get("litellm_params")
if not isinstance(litellm_params, dict):
return
metadata = litellm_params.get("metadata")
if not isinstance(metadata, dict):
return
otel_internal = metadata.get("_otel_internal")
if not isinstance(otel_internal, dict):
return
spans_logged = otel_internal.get("spans_logged")
if not isinstance(spans_logged, dict):
return
for scope in [key for key in spans_logged if isinstance(key, tuple) and key[-1:] == ("success",)]:
del spans_logged[scope]
def _flush_passthrough_collected_chunks_helper(
self,
raw_bytes: list[bytes],

View file

@ -478,9 +478,12 @@ def _write_spend_metadata_to_kwargs(kwargs: dict, metadata: dict) -> None:
def _is_unbilled_in_progress_interaction(completion_response: object) -> bool:
from litellm.interactions.background_cost_polling import is_pollable_background_interaction
from litellm.types.interactions import InteractionsAPIResponse
return isinstance(completion_response, InteractionsAPIResponse) and completion_response.usage is None
if not isinstance(completion_response, InteractionsAPIResponse):
return False
return completion_response.usage is None and is_pollable_background_interaction(completion_response)
def _should_track_cost_callback(

View file

@ -4644,6 +4644,44 @@ async def test_background_interaction_completion_rebills_after_in_progress_succe
assert logging_obj.model_call_details["standard_logging_object"]["total_tokens"] == 175
@pytest.mark.asyncio
async def test_background_interaction_completion_lets_otel_emit_the_cost_span():
"""
OTEL, and every integration that derives from it, dedupes span emission on
a marker kept in the request's own metadata. The in-progress create claims
that marker, so without clearing it the settled completion, the only event
carrying usage and cost, is discarded as a duplicate and every
OTEL-family backend shows the interaction as a span with no cost at all.
"""
import datetime as dt
from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig
from litellm.types.interactions import InteractionsAPIResponse
otel = OpenTelemetry(config=OpenTelemetryConfig(exporter="console"))
logging_obj = _interactions_logging_obj(stream=False)
in_progress = InteractionsAPIResponse(id="interactions/abc", model="gemini-2.5-flash", status="in_progress")
await logging_obj.async_success_handler(
result=in_progress,
start_time=dt.datetime.now(),
end_time=dt.datetime.now(),
)
assert otel._emit_once(logging_obj.model_call_details, "success") is True
assert otel._emit_once(logging_obj.model_call_details, "success") is False
completed = InteractionsAPIResponse(
id="interactions/abc",
model="gemini-2.5-flash",
status="completed",
steps=[],
usage=dict(INTERACTIONS_USAGE_BLOCK),
)
await logging_obj.async_log_background_interaction_completion(result=completed)
assert otel._emit_once(logging_obj.model_call_details, "success") is True
@pytest.mark.parametrize(
"call_type",
["aget", "get", "aget_interaction", "adelete_interaction", "acancel_interaction"],

View file

@ -859,6 +859,105 @@ async def test_track_cost_callback_releases_reservation_for_in_progress_interact
mock_proxy_logging.failed_tracking_alert.assert_not_called()
@pytest.mark.asyncio
@pytest.mark.parametrize(
"status",
["completed", "failed", "cancelled", "incomplete", "requires_action", "budget_exceeded"],
)
async def test_track_cost_callback_releases_reservation_for_unpollable_interaction(status):
"""
Only an in-progress create gets a poll task, so a create that comes back
terminal with no usage has nobody left to reconcile its reservation. The
callback must release it there and then, or the pre-call estimate stays
added to the key, user, team and org spend counters and starts refusing
traffic against budget that was never actually spent.
"""
from litellm.types.interactions import InteractionsAPIResponse
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
terminal_response = InteractionsAPIResponse(
id="interactions/bg-abc",
model="gemini-3-flash-preview",
status=status,
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=terminal_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert reservation["finalized"] is True
@pytest.mark.asyncio
async def test_track_cost_callback_releases_reservation_for_interaction_without_an_id():
"""
The scheduler also refuses a response with no id, since it has nothing to
poll for, so the callback must not defer to a poll task that will never
exist.
"""
from litellm.types.interactions import InteractionsAPIResponse
logger = _ProxyDBLogger()
reservation = {"reserved_cost": 0.05, "entries": [], "finalized": False}
idless_response = InteractionsAPIResponse(
id="",
model="gemini-3-flash-preview",
status="in_progress",
)
with patch(
"litellm.proxy.proxy_server.proxy_logging_obj",
) as mock_proxy_logging:
mock_proxy_logging.failed_tracking_alert = AsyncMock()
mock_proxy_logging.db_spend_update_writer = MagicMock()
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
await logger._PROXY_track_cost_callback(
kwargs=_in_progress_interaction_kwargs(reservation),
completion_response=idless_response,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert reservation["finalized"] is True
@pytest.mark.parametrize(
"status",
["in_progress", "completed", "failed", "cancelled", "incomplete", "requires_action"],
)
@pytest.mark.parametrize("interaction_id", ["interactions/bg-abc", ""])
def test_callback_defers_exactly_the_interactions_the_scheduler_polls(status, interaction_id):
"""
Pins the invariant the two modules share: the callback may only hold a
budget reservation open for a response the scheduler will actually poll.
Any drift between the two gates leaks reservations onto live spend
counters, so assert they agree rather than restating either condition.
"""
from litellm.interactions.background_cost_polling import is_pollable_background_interaction
from litellm.proxy.hooks.proxy_track_cost_callback import _is_unbilled_in_progress_interaction
from litellm.types.interactions import InteractionsAPIResponse
response = InteractionsAPIResponse(
id=interaction_id,
model="gemini-3-flash-preview",
status=status,
)
assert _is_unbilled_in_progress_interaction(response) is is_pollable_background_interaction(response)
@pytest.mark.asyncio
async def test_async_post_call_failure_hook_propagates_trace_id_from_logging_obj():
"""