fix(proxy): use Responses-API transformer in pass-through cost tracking (#29728)

The `elif is_responses:` branch of `openai_passthrough_handler` was
calling the chat-completions `transform_response` on a Responses API
payload. The chat-completions transformer expects `choices: [...]`
in the raw response; the Responses API uses `output: [...]` and
`usage.input_tokens` / `usage.output_tokens` (not
`prompt_tokens` / `completion_tokens`). The result was a
KeyError 'choices' deep inside `convert_to_model_response_object`,
swallowed by the surrounding `except Exception` in the handler, and
the SpendLogs row was written by the fallback path with zeroed-out
tokens, spend, and model.

This bug silently undercounts cost for every successful pass-through
call to either OpenAI's `/v1/responses` or Azure's
`/openai/v1/responses` (deployments configured for the Responses
API). Reproduced 2026-06-04 against a real Azure OpenAI Responses
API deployment proxied through LiteLLM v1.88.0.

Fix: use the dedicated
`OpenAIResponsesAPIConfig.transform_response_api_response` for the
Responses branch. This transformer already exists in LiteLLM
(`litellm/llms/openai/responses/transformation.py`) and knows the
Responses-API on-the-wire shape. `litellm.completion_cost` already
handles `ResponsesAPIResponse` natively with `call_type="responses"`,
so no downstream changes are needed.

Tests:

  test_responses_api_uses_responses_transformer_not_chat_completions
    NEW. Real regression test — exercises the openai_passthrough_handler
    with a real-shaped Responses payload (no `choices`, has `output`
    and Responses-API `usage` keys) and NO mocked `get_provider_config`.
    Pre-fix: raises KeyError 'choices' inside the chat-completions
    transformer (the bug). Post-fix: returns a ResponsesAPIResponse,
    completion_cost is called with call_type="responses" and a
    ResponsesAPIResponse instance (asserted).
    Verified to fail on un-fixed handler + pass on fixed handler
    before commit.

  test_responses_api_cost_tracking
    UPDATED. Old test mocked `get_provider_config` (no longer called
    in the responses branch post-fix). Now mocks the Responses
    transformer directly (`OpenAIResponsesAPIConfig.transform_response_api_response`)
    to test the downstream cost-calc contract.

Out of scope for this PR (separate followup):
  - Recognizing *.cognitiveservices.azure.com (the newer Azure
    OpenAI hostname) in the is_openai_*_route checks. Separate PR.

Co-authored-by: shin-berri <shin-laptop@berri.ai>
Co-authored-by: yuneng-jiang <yuneng@berri.ai>
This commit is contained in:
Liam Scott 2026-06-10 06:16:14 -04:00 • committed by Sameer Kankute
parent 7729ff5a13
commit 217cedc988
No known key found for this signature in database
4 changed files with 315 additions and 42 deletions

View file

@ -3834,6 +3834,7 @@ PassThroughEndpointLoggingResultValues = Union[
EmbeddingResponse,
VideoObject,
StandardPassThroughResponseObject,
ResponsesAPIResponse,
]

View file

@ -5,7 +5,7 @@ Handles cost tracking and logging for OpenAI passthrough endpoints, specifically
"""
from datetime import datetime
from typing import List, Optional, Union
from typing import List, Optional, Tuple, Union
from urllib.parse import urlparse
import httpx
@ -18,6 +18,7 @@ from litellm.litellm_core_utils.litellm_logging import (
)
from litellm.llms.openai.openai import OpenAIConfig
from litellm.llms.openai.openai import OpenAIConfig as OpenAIConfigType
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
from litellm.proxy._types import PassThroughEndpointLoggingTypedDict
from litellm.proxy.pass_through_endpoints.llm_provider_handlers.base_passthrough_logging_handler import (
BasePassthroughLoggingHandler,
@ -29,6 +30,7 @@ from litellm.types.passthrough_endpoints.pass_through_endpoints import (
EndpointType,
PassthroughStandardLoggingPayload,
)
from litellm.types.llms.openai import ResponsesAPIResponse
from litellm.types.utils import ImageResponse, LlmProviders, PassthroughCallTypes
from litellm.utils import ModelResponse, TextCompletionResponse
@ -236,6 +238,42 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
)
return 0.0
@staticmethod
def _build_responses_api_response_and_cost(
model: str,
httpx_response: httpx.Response,
logging_obj: LiteLLMLoggingObj,
custom_llm_provider: str,
) -> Tuple[ResponsesAPIResponse, float]:
"""Transform a Responses API raw response into a ResponsesAPIResponse
and compute its cost.
The Responses API has a different on-the-wire shape from chat
completions (`output: [...]` instead of `choices: [...]`), so the
chat-completions `transform_response` raises KeyError 'choices' on
a Responses payload. Use the dedicated Responses-API transformer
(`OpenAIResponsesAPIConfig.transform_response_api_response`) here.
Returns (litellm_model_response, response_cost) — symmetric with the
chat-completions branch which produces the same two values inline,
and analogous to the image branches' `_calculate_image_*_cost` helpers
(which return cost only because the image-response object is trivial
to build inline; the Responses payload needs a real transformer).
"""
responses_config = OpenAIResponsesAPIConfig()
litellm_model_response = responses_config.transform_response_api_response(
model=model,
raw_response=httpx_response,
logging_obj=logging_obj,
)
response_cost = litellm.completion_cost(
completion_response=litellm_model_response,
model=model,
custom_llm_provider=custom_llm_provider,
call_type="responses",
)
return litellm_model_response, response_cost
@staticmethod
def openai_passthrough_handler( # noqa: PLR0915
httpx_response: httpx.Response,
@ -301,7 +339,12 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
try:
response_cost = 0.0
litellm_model_response: Optional[
Union[ModelResponse, TextCompletionResponse, ImageResponse]
Union[
ModelResponse,
TextCompletionResponse,
ImageResponse,
ResponsesAPIResponse,
]
] = None
handler_instance = OpenAIPassthroughLoggingHandler()
@ -384,29 +427,18 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
litellm_model_response._hidden_params = {}
litellm_model_response._hidden_params["response_cost"] = response_cost
elif is_responses:
# Handle responses API cost calculation
provider_config = handler_instance.get_provider_config(model=model)
existing_litellm_params = kwargs.get("litellm_params", {}) or {}
litellm_model_response = provider_config.transform_response(
raw_response=httpx_response,
model_response=litellm.ModelResponse(),
# Responses-API cost tracking — see
# `_build_responses_api_response_and_cost` for why this needs
# a dedicated transformer (the chat-completions transform
# crashes on the Responses payload shape).
(
litellm_model_response,
response_cost,
) = OpenAIPassthroughLoggingHandler._build_responses_api_response_and_cost(
model=model,
messages=request_body.get("messages", []),
httpx_response=httpx_response,
logging_obj=logging_obj,
optional_params=request_body.get("optional_params", {}),
api_key="",
request_data=request_body,
encoding=litellm.encoding,
json_mode=False,
litellm_params=existing_litellm_params,
)
# Calculate cost using LiteLLM's cost calculator with responses call type
response_cost = litellm.completion_cost(
completion_response=litellm_model_response,
model=model,
custom_llm_provider=custom_llm_provider,
call_type="responses",
)
# Update kwargs with cost information

View file

@ -458,7 +458,16 @@ class PassThroughEndpointLogging:
return False
def _is_supported_openai_endpoint(self, url_route: str) -> bool:
"""Check if the OpenAI endpoint is supported by the passthrough logging handler."""
"""Check if the OpenAI endpoint is supported by the passthrough logging handler.
The Responses API route is included because
`openai_passthrough_handler` has a dedicated `elif is_responses:`
branch that knows how to extract usage + cost from the
Responses-API on-the-wire shape. Without including it here, the
outer dispatch filters Responses calls out before reaching the
handler — the inner branch is then unreachable and Responses
calls land in `LiteLLM_SpendLogs` with zero tokens / zero spend.
"""
from .llm_provider_handlers.openai_passthrough_logging_handler import (
OpenAIPassthroughLoggingHandler,
)
@ -469,6 +478,7 @@ class PassThroughEndpointLogging:
url_route
)
or OpenAIPassthroughLoggingHandler.is_openai_image_editing_route(url_route)
or OpenAIPassthroughLoggingHandler.is_openai_responses_route(url_route)
)
def _set_cost_per_request(

View file

@ -683,36 +683,43 @@ class TestOpenAIPassthroughLoggingHandler:
"litellm.litellm_core_utils.litellm_logging.get_standard_logging_object_payload"
)
@patch(
"litellm.proxy.pass_through_endpoints.llm_provider_handlers.openai_passthrough_logging_handler.OpenAIPassthroughLoggingHandler.get_provider_config"
"litellm.llms.openai.responses.transformation.OpenAIResponsesAPIConfig.transform_response_api_response"
)
def test_responses_api_cost_tracking(
self, mock_get_provider_config, mock_get_standard_logging, mock_completion_cost
self,
mock_transform_responses,
mock_get_standard_logging,
mock_completion_cost,
):
"""Test cost tracking for responses API route"""
"""Test cost tracking for responses API route.
Mocks the Responses-API transformer (the dedicated one this branch
of the handler dispatches into post-fix) so we can assert the
downstream cost-calculation contract without depending on the
real transformer's full behavior.
"""
# Arrange
mock_completion_cost.return_value = 0.000050
mock_get_standard_logging.return_value = {"test": "logging_payload"}
# Mock the provider config's transform_response to return a valid ModelResponse
from litellm import ModelResponse
# Mock the Responses transformer's return — a ResponsesAPIResponse
# carrying the usage fields downstream cost-calc expects.
from litellm.types.llms.openai import ResponsesAPIResponse
mock_model_response = ModelResponse(
mock_responses_api_response = ResponsesAPIResponse.model_construct(
id="resp_abc123",
object="response",
created_at=1677652288,
model="gpt-4o-2024-08-06",
choices=[
{
"message": {
"role": "assistant",
"content": "Hello! How can I help you today?",
}
}
],
usage={"prompt_tokens": 20, "completion_tokens": 15, "total_tokens": 35},
status="completed",
output=[],
usage={
"input_tokens": 20,
"output_tokens": 15,
"total_tokens": 35,
},
)
mock_provider_config = MagicMock()
mock_provider_config.transform_response.return_value = mock_model_response
mock_get_provider_config.return_value = mock_provider_config
mock_transform_responses.return_value = mock_responses_api_response
# Mock responses API response
mock_responses_response = {
@ -768,6 +775,109 @@ class TestOpenAIPassthroughLoggingHandler:
assert mock_logging_obj.model_call_details["model"] == "gpt-4o"
assert mock_logging_obj.model_call_details["custom_llm_provider"] == "openai"
@patch("litellm.completion_cost")
@patch(
"litellm.litellm_core_utils.litellm_logging.get_standard_logging_object_payload"
)
def test_responses_api_uses_responses_transformer_not_chat_completions(
self, mock_get_standard_logging, mock_completion_cost
):
"""Regression test for the Responses-API cost-tracking dispatch bug.
BUG: the `elif is_responses:` branch in `openai_passthrough_handler`
was calling `OpenAIConfig.transform_response` (the chat-completions
transformer) on a Responses API payload. Chat-completions
transform_response expects `choices: [...]` in the raw response;
the Responses API uses `output: [...]` and `usage.input_tokens` /
`usage.output_tokens` (not `prompt_tokens` / `completion_tokens`).
The result was a KeyError 'choices' inside
`convert_to_model_response_object`, swallowed by the surrounding
try/except, and the SpendLogs row was written with zero tokens
and zero spend.
FIX: use the dedicated `OpenAIResponsesAPIConfig.transform_response_api_response`
for the Responses branch.
This test exercises the REAL transformer (no mocked
`get_provider_config`) so that running it against the un-fixed
handler raises and running it against the fixed handler succeeds.
"""
mock_completion_cost.return_value = 0.000050
mock_get_standard_logging.return_value = {"test": "logging_payload"}
# A real-shaped Azure / OpenAI Responses API payload — NO `choices`,
# uses `output` and `usage.input_tokens` / `usage.output_tokens`.
responses_api_body = {
"id": "resp_abc123",
"object": "response",
"created_at": 1677652288,
"model": "gpt-4o-2024-08-06",
"status": "completed",
"output": [
{
"type": "message",
"role": "assistant",
"content": [
{
"type": "output_text",
"text": "Hello!",
}
],
}
],
"usage": {
"input_tokens": 20,
"output_tokens": 15,
"total_tokens": 35,
},
}
mock_httpx_response = self._create_mock_httpx_response(responses_api_body)
mock_logging_obj = self._create_mock_logging_obj()
passthrough_payload = self._create_passthrough_logging_payload()
kwargs = {
"passthrough_logging_payload": passthrough_payload,
"model": "gpt-4o",
"custom_llm_provider": "openai",
}
result = OpenAIPassthroughLoggingHandler.openai_passthrough_handler(
httpx_response=mock_httpx_response,
response_body=responses_api_body,
logging_obj=mock_logging_obj,
url_route="https://api.openai.com/v1/responses",
result="",
start_time=self.start_time,
end_time=self.end_time,
cache_hit=False,
request_body={"model": "gpt-4o", "input": "Tell me about AI"},
**kwargs,
)
# Pre-fix this assertion fails — the handler swallows the
# KeyError raised by the chat-completions transformer and falls
# back to the passthrough_chat_handler which yields a different
# response_cost value. Post-fix, the Responses transformer
# succeeds and we get the mocked 0.000050.
assert result is not None
assert result["kwargs"]["response_cost"] == 0.000050
assert result["kwargs"]["model"] == "gpt-4o"
# `completion_cost` must be called with the responses call type
# and a `ResponsesAPIResponse` (not a `ModelResponse`).
mock_completion_cost.assert_called_once()
call_kwargs = mock_completion_cost.call_args[1]
assert call_kwargs["call_type"] == "responses"
from litellm.types.llms.openai import ResponsesAPIResponse
assert isinstance(call_kwargs["completion_response"], ResponsesAPIResponse), (
"completion_response must be a ResponsesAPIResponse; passing a "
"chat-completions ModelResponse means the Responses transformer "
"isn't being used and we're back in the bug."
)
class TestOpenAIPassthroughIntegration:
"""Integration tests for OpenAI passthrough cost tracking"""
@ -872,6 +982,126 @@ class TestOpenAIPassthroughIntegration:
)
assert self.handler.is_openai_route("") == False
def test_is_supported_openai_endpoint_includes_responses_api(self):
"""Regression test for the outer dispatch gate.
`_is_supported_openai_endpoint` is the gate that decides whether the
OpenAI handler runs for a given URL. Before this gate accepted the
Responses API, calls to `/v1/responses` would fail the gate and the
handler's `elif is_responses:` branch was unreachable in the live
success-handler pipeline — every Responses-API call landed in
`LiteLLM_SpendLogs` with zero tokens / zero spend even though the
handler had a Responses branch internally.
This test exercises the dispatch decision directly so future
refactors of `_is_supported_openai_endpoint` can't silently
remove Responses from the OR-chain without a test failure.
"""
# Responses must be supported on api.openai.com and openai.azure.com.
assert (
self.handler._is_supported_openai_endpoint(
"https://api.openai.com/v1/responses"
)
is True
)
assert (
self.handler._is_supported_openai_endpoint(
"https://openai.azure.com/v1/responses"
)
is True
)
# The other supported endpoints stay supported (no regression).
assert (
self.handler._is_supported_openai_endpoint(
"https://api.openai.com/v1/chat/completions"
)
is True
)
assert (
self.handler._is_supported_openai_endpoint(
"https://api.openai.com/v1/images/generations"
)
is True
)
assert (
self.handler._is_supported_openai_endpoint(
"https://api.openai.com/v1/images/edits"
)
is True
)
# Unsupported OpenAI endpoints (e.g. /v1/models) still return False.
assert (
self.handler._is_supported_openai_endpoint(
"https://api.openai.com/v1/models"
)
is False
)
@patch(
"litellm.proxy.pass_through_endpoints.llm_provider_handlers.openai_passthrough_logging_handler.OpenAIPassthroughLoggingHandler.openai_passthrough_handler"
)
@pytest.mark.asyncio
async def test_success_handler_dispatches_responses_api_to_openai_handler(
self, mock_openai_handler
):
"""End-to-end dispatch test for the Responses API path.
Pre-fix: `_is_supported_openai_endpoint` returned False for
`/v1/responses` URLs, so the OpenAI handler was never called.
This test would fail (mock never invoked) on the un-fixed
success_handler — passes only when the dispatch gate accepts
Responses URLs.
"""
mock_openai_handler.return_value = {
"result": {"id": "resp_abc123"},
"kwargs": {
"response_cost": 0.0001,
"model": "gpt-4o",
"custom_llm_provider": "openai",
},
}
mock_httpx_response = MagicMock(spec=httpx.Response)
mock_httpx_response.text = (
'{"id": "resp_abc123", "object": "response", '
'"output": [], "usage": {"input_tokens": 5, "output_tokens": 3}}'
)
mock_logging_obj = AsyncMock()
mock_logging_obj.model_call_details = {}
mock_logging_obj.async_success_handler = AsyncMock()
passthrough_payload = PassthroughStandardLoggingPayload(
url="https://api.openai.com/v1/responses",
request_body={"model": "gpt-4o", "input": "Hello"},
request_method="POST",
)
await self.handler.pass_through_async_success_handler(
httpx_response=mock_httpx_response,
response_body={
"id": "resp_abc123",
"object": "response",
"output": [],
"usage": {"input_tokens": 5, "output_tokens": 3},
},
logging_obj=mock_logging_obj,
url_route="https://api.openai.com/v1/responses",
result="",
start_time=datetime.now(),
end_time=datetime.now(),
cache_hit=False,
request_body={"model": "gpt-4o", "input": "Hello"},
passthrough_logging_payload=passthrough_payload,
)
# The OpenAI handler MUST have been invoked. Pre-fix the dispatch
# gate filtered Responses URLs out and the mock was never called.
mock_openai_handler.assert_called_once()
# And we can verify it was dispatched with the Responses URL.
call_kwargs = mock_openai_handler.call_args.kwargs
assert call_kwargs["url_route"] == "https://api.openai.com/v1/responses"
@patch(
"litellm.proxy.pass_through_endpoints.llm_provider_handlers.openai_passthrough_logging_handler.OpenAIPassthroughLoggingHandler.openai_passthrough_handler"
)