mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(proxy): use Responses-API transformer in pass-through cost tracking (#29728)
The `elif is_responses:` branch of `openai_passthrough_handler` was
calling the chat-completions `transform_response` on a Responses API
payload. The chat-completions transformer expects `choices: [...]`
in the raw response; the Responses API uses `output: [...]` and
`usage.input_tokens` / `usage.output_tokens` (not
`prompt_tokens` / `completion_tokens`). The result was a
KeyError 'choices' deep inside `convert_to_model_response_object`,
swallowed by the surrounding `except Exception` in the handler, and
the SpendLogs row was written by the fallback path with zeroed-out
tokens, spend, and model.
This bug silently undercounts cost for every successful pass-through
call to either OpenAI's `/v1/responses` or Azure's
`/openai/v1/responses` (deployments configured for the Responses
API). Reproduced 2026-06-04 against a real Azure OpenAI Responses
API deployment proxied through LiteLLM v1.88.0.
Fix: use the dedicated
`OpenAIResponsesAPIConfig.transform_response_api_response` for the
Responses branch. This transformer already exists in LiteLLM
(`litellm/llms/openai/responses/transformation.py`) and knows the
Responses-API on-the-wire shape. `litellm.completion_cost` already
handles `ResponsesAPIResponse` natively with `call_type="responses"`,
so no downstream changes are needed.
Tests:
test_responses_api_uses_responses_transformer_not_chat_completions
NEW. Real regression test — exercises the openai_passthrough_handler
with a real-shaped Responses payload (no `choices`, has `output`
and Responses-API `usage` keys) and NO mocked `get_provider_config`.
Pre-fix: raises KeyError 'choices' inside the chat-completions
transformer (the bug). Post-fix: returns a ResponsesAPIResponse,
completion_cost is called with call_type="responses" and a
ResponsesAPIResponse instance (asserted).
Verified to fail on un-fixed handler + pass on fixed handler
before commit.
test_responses_api_cost_tracking
UPDATED. Old test mocked `get_provider_config` (no longer called
in the responses branch post-fix). Now mocks the Responses
transformer directly (`OpenAIResponsesAPIConfig.transform_response_api_response`)
to test the downstream cost-calc contract.
Out of scope for this PR (separate followup):
- Recognizing *.cognitiveservices.azure.com (the newer Azure
OpenAI hostname) in the is_openai_*_route checks. Separate PR.
Co-authored-by: shin-berri <shin-laptop@berri.ai>
Co-authored-by: yuneng-jiang <yuneng@berri.ai>
This commit is contained in:
parent
7729ff5a13
commit
217cedc988
4 changed files with 315 additions and 42 deletions
|
|
@ -3834,6 +3834,7 @@ PassThroughEndpointLoggingResultValues = Union[
|
|||
EmbeddingResponse,
|
||||
VideoObject,
|
||||
StandardPassThroughResponseObject,
|
||||
ResponsesAPIResponse,
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ Handles cost tracking and logging for OpenAI passthrough endpoints, specifically
|
|||
"""
|
||||
|
||||
from datetime import datetime
|
||||
from typing import List, Optional, Union
|
||||
from typing import List, Optional, Tuple, Union
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
|
|
@ -18,6 +18,7 @@ from litellm.litellm_core_utils.litellm_logging import (
|
|||
)
|
||||
from litellm.llms.openai.openai import OpenAIConfig
|
||||
from litellm.llms.openai.openai import OpenAIConfig as OpenAIConfigType
|
||||
from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig
|
||||
from litellm.proxy._types import PassThroughEndpointLoggingTypedDict
|
||||
from litellm.proxy.pass_through_endpoints.llm_provider_handlers.base_passthrough_logging_handler import (
|
||||
BasePassthroughLoggingHandler,
|
||||
|
|
@ -29,6 +30,7 @@ from litellm.types.passthrough_endpoints.pass_through_endpoints import (
|
|||
EndpointType,
|
||||
PassthroughStandardLoggingPayload,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
from litellm.types.utils import ImageResponse, LlmProviders, PassthroughCallTypes
|
||||
from litellm.utils import ModelResponse, TextCompletionResponse
|
||||
|
||||
|
|
@ -236,6 +238,42 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
|
|||
)
|
||||
return 0.0
|
||||
|
||||
@staticmethod
|
||||
def _build_responses_api_response_and_cost(
|
||||
model: str,
|
||||
httpx_response: httpx.Response,
|
||||
logging_obj: LiteLLMLoggingObj,
|
||||
custom_llm_provider: str,
|
||||
) -> Tuple[ResponsesAPIResponse, float]:
|
||||
"""Transform a Responses API raw response into a ResponsesAPIResponse
|
||||
and compute its cost.
|
||||
|
||||
The Responses API has a different on-the-wire shape from chat
|
||||
completions (`output: [...]` instead of `choices: [...]`), so the
|
||||
chat-completions `transform_response` raises KeyError 'choices' on
|
||||
a Responses payload. Use the dedicated Responses-API transformer
|
||||
(`OpenAIResponsesAPIConfig.transform_response_api_response`) here.
|
||||
|
||||
Returns (litellm_model_response, response_cost) — symmetric with the
|
||||
chat-completions branch which produces the same two values inline,
|
||||
and analogous to the image branches' `_calculate_image_*_cost` helpers
|
||||
(which return cost only because the image-response object is trivial
|
||||
to build inline; the Responses payload needs a real transformer).
|
||||
"""
|
||||
responses_config = OpenAIResponsesAPIConfig()
|
||||
litellm_model_response = responses_config.transform_response_api_response(
|
||||
model=model,
|
||||
raw_response=httpx_response,
|
||||
logging_obj=logging_obj,
|
||||
)
|
||||
response_cost = litellm.completion_cost(
|
||||
completion_response=litellm_model_response,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
call_type="responses",
|
||||
)
|
||||
return litellm_model_response, response_cost
|
||||
|
||||
@staticmethod
|
||||
def openai_passthrough_handler( # noqa: PLR0915
|
||||
httpx_response: httpx.Response,
|
||||
|
|
@ -301,7 +339,12 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
|
|||
try:
|
||||
response_cost = 0.0
|
||||
litellm_model_response: Optional[
|
||||
Union[ModelResponse, TextCompletionResponse, ImageResponse]
|
||||
Union[
|
||||
ModelResponse,
|
||||
TextCompletionResponse,
|
||||
ImageResponse,
|
||||
ResponsesAPIResponse,
|
||||
]
|
||||
] = None
|
||||
handler_instance = OpenAIPassthroughLoggingHandler()
|
||||
|
||||
|
|
@ -384,29 +427,18 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
|
|||
litellm_model_response._hidden_params = {}
|
||||
litellm_model_response._hidden_params["response_cost"] = response_cost
|
||||
elif is_responses:
|
||||
# Handle responses API cost calculation
|
||||
provider_config = handler_instance.get_provider_config(model=model)
|
||||
existing_litellm_params = kwargs.get("litellm_params", {}) or {}
|
||||
litellm_model_response = provider_config.transform_response(
|
||||
raw_response=httpx_response,
|
||||
model_response=litellm.ModelResponse(),
|
||||
# Responses-API cost tracking — see
|
||||
# `_build_responses_api_response_and_cost` for why this needs
|
||||
# a dedicated transformer (the chat-completions transform
|
||||
# crashes on the Responses payload shape).
|
||||
(
|
||||
litellm_model_response,
|
||||
response_cost,
|
||||
) = OpenAIPassthroughLoggingHandler._build_responses_api_response_and_cost(
|
||||
model=model,
|
||||
messages=request_body.get("messages", []),
|
||||
httpx_response=httpx_response,
|
||||
logging_obj=logging_obj,
|
||||
optional_params=request_body.get("optional_params", {}),
|
||||
api_key="",
|
||||
request_data=request_body,
|
||||
encoding=litellm.encoding,
|
||||
json_mode=False,
|
||||
litellm_params=existing_litellm_params,
|
||||
)
|
||||
|
||||
# Calculate cost using LiteLLM's cost calculator with responses call type
|
||||
response_cost = litellm.completion_cost(
|
||||
completion_response=litellm_model_response,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
call_type="responses",
|
||||
)
|
||||
|
||||
# Update kwargs with cost information
|
||||
|
|
|
|||
|
|
@ -458,7 +458,16 @@ class PassThroughEndpointLogging:
|
|||
return False
|
||||
|
||||
def _is_supported_openai_endpoint(self, url_route: str) -> bool:
|
||||
"""Check if the OpenAI endpoint is supported by the passthrough logging handler."""
|
||||
"""Check if the OpenAI endpoint is supported by the passthrough logging handler.
|
||||
|
||||
The Responses API route is included because
|
||||
`openai_passthrough_handler` has a dedicated `elif is_responses:`
|
||||
branch that knows how to extract usage + cost from the
|
||||
Responses-API on-the-wire shape. Without including it here, the
|
||||
outer dispatch filters Responses calls out before reaching the
|
||||
handler — the inner branch is then unreachable and Responses
|
||||
calls land in `LiteLLM_SpendLogs` with zero tokens / zero spend.
|
||||
"""
|
||||
from .llm_provider_handlers.openai_passthrough_logging_handler import (
|
||||
OpenAIPassthroughLoggingHandler,
|
||||
)
|
||||
|
|
@ -469,6 +478,7 @@ class PassThroughEndpointLogging:
|
|||
url_route
|
||||
)
|
||||
or OpenAIPassthroughLoggingHandler.is_openai_image_editing_route(url_route)
|
||||
or OpenAIPassthroughLoggingHandler.is_openai_responses_route(url_route)
|
||||
)
|
||||
|
||||
def _set_cost_per_request(
|
||||
|
|
|
|||
|
|
@ -683,36 +683,43 @@ class TestOpenAIPassthroughLoggingHandler:
|
|||
"litellm.litellm_core_utils.litellm_logging.get_standard_logging_object_payload"
|
||||
)
|
||||
@patch(
|
||||
"litellm.proxy.pass_through_endpoints.llm_provider_handlers.openai_passthrough_logging_handler.OpenAIPassthroughLoggingHandler.get_provider_config"
|
||||
"litellm.llms.openai.responses.transformation.OpenAIResponsesAPIConfig.transform_response_api_response"
|
||||
)
|
||||
def test_responses_api_cost_tracking(
|
||||
self, mock_get_provider_config, mock_get_standard_logging, mock_completion_cost
|
||||
self,
|
||||
mock_transform_responses,
|
||||
mock_get_standard_logging,
|
||||
mock_completion_cost,
|
||||
):
|
||||
"""Test cost tracking for responses API route"""
|
||||
"""Test cost tracking for responses API route.
|
||||
|
||||
Mocks the Responses-API transformer (the dedicated one this branch
|
||||
of the handler dispatches into post-fix) so we can assert the
|
||||
downstream cost-calculation contract without depending on the
|
||||
real transformer's full behavior.
|
||||
"""
|
||||
# Arrange
|
||||
mock_completion_cost.return_value = 0.000050
|
||||
mock_get_standard_logging.return_value = {"test": "logging_payload"}
|
||||
|
||||
# Mock the provider config's transform_response to return a valid ModelResponse
|
||||
from litellm import ModelResponse
|
||||
# Mock the Responses transformer's return — a ResponsesAPIResponse
|
||||
# carrying the usage fields downstream cost-calc expects.
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
mock_model_response = ModelResponse(
|
||||
mock_responses_api_response = ResponsesAPIResponse.model_construct(
|
||||
id="resp_abc123",
|
||||
object="response",
|
||||
created_at=1677652288,
|
||||
model="gpt-4o-2024-08-06",
|
||||
choices=[
|
||||
{
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "Hello! How can I help you today?",
|
||||
}
|
||||
}
|
||||
],
|
||||
usage={"prompt_tokens": 20, "completion_tokens": 15, "total_tokens": 35},
|
||||
status="completed",
|
||||
output=[],
|
||||
usage={
|
||||
"input_tokens": 20,
|
||||
"output_tokens": 15,
|
||||
"total_tokens": 35,
|
||||
},
|
||||
)
|
||||
|
||||
mock_provider_config = MagicMock()
|
||||
mock_provider_config.transform_response.return_value = mock_model_response
|
||||
mock_get_provider_config.return_value = mock_provider_config
|
||||
mock_transform_responses.return_value = mock_responses_api_response
|
||||
|
||||
# Mock responses API response
|
||||
mock_responses_response = {
|
||||
|
|
@ -768,6 +775,109 @@ class TestOpenAIPassthroughLoggingHandler:
|
|||
assert mock_logging_obj.model_call_details["model"] == "gpt-4o"
|
||||
assert mock_logging_obj.model_call_details["custom_llm_provider"] == "openai"
|
||||
|
||||
@patch("litellm.completion_cost")
|
||||
@patch(
|
||||
"litellm.litellm_core_utils.litellm_logging.get_standard_logging_object_payload"
|
||||
)
|
||||
def test_responses_api_uses_responses_transformer_not_chat_completions(
|
||||
self, mock_get_standard_logging, mock_completion_cost
|
||||
):
|
||||
"""Regression test for the Responses-API cost-tracking dispatch bug.
|
||||
|
||||
BUG: the `elif is_responses:` branch in `openai_passthrough_handler`
|
||||
was calling `OpenAIConfig.transform_response` (the chat-completions
|
||||
transformer) on a Responses API payload. Chat-completions
|
||||
transform_response expects `choices: [...]` in the raw response;
|
||||
the Responses API uses `output: [...]` and `usage.input_tokens` /
|
||||
`usage.output_tokens` (not `prompt_tokens` / `completion_tokens`).
|
||||
The result was a KeyError 'choices' inside
|
||||
`convert_to_model_response_object`, swallowed by the surrounding
|
||||
try/except, and the SpendLogs row was written with zero tokens
|
||||
and zero spend.
|
||||
|
||||
FIX: use the dedicated `OpenAIResponsesAPIConfig.transform_response_api_response`
|
||||
for the Responses branch.
|
||||
|
||||
This test exercises the REAL transformer (no mocked
|
||||
`get_provider_config`) so that running it against the un-fixed
|
||||
handler raises and running it against the fixed handler succeeds.
|
||||
"""
|
||||
mock_completion_cost.return_value = 0.000050
|
||||
mock_get_standard_logging.return_value = {"test": "logging_payload"}
|
||||
|
||||
# A real-shaped Azure / OpenAI Responses API payload — NO `choices`,
|
||||
# uses `output` and `usage.input_tokens` / `usage.output_tokens`.
|
||||
responses_api_body = {
|
||||
"id": "resp_abc123",
|
||||
"object": "response",
|
||||
"created_at": 1677652288,
|
||||
"model": "gpt-4o-2024-08-06",
|
||||
"status": "completed",
|
||||
"output": [
|
||||
{
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{
|
||||
"type": "output_text",
|
||||
"text": "Hello!",
|
||||
}
|
||||
],
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 20,
|
||||
"output_tokens": 15,
|
||||
"total_tokens": 35,
|
||||
},
|
||||
}
|
||||
|
||||
mock_httpx_response = self._create_mock_httpx_response(responses_api_body)
|
||||
mock_logging_obj = self._create_mock_logging_obj()
|
||||
passthrough_payload = self._create_passthrough_logging_payload()
|
||||
|
||||
kwargs = {
|
||||
"passthrough_logging_payload": passthrough_payload,
|
||||
"model": "gpt-4o",
|
||||
"custom_llm_provider": "openai",
|
||||
}
|
||||
|
||||
result = OpenAIPassthroughLoggingHandler.openai_passthrough_handler(
|
||||
httpx_response=mock_httpx_response,
|
||||
response_body=responses_api_body,
|
||||
logging_obj=mock_logging_obj,
|
||||
url_route="https://api.openai.com/v1/responses",
|
||||
result="",
|
||||
start_time=self.start_time,
|
||||
end_time=self.end_time,
|
||||
cache_hit=False,
|
||||
request_body={"model": "gpt-4o", "input": "Tell me about AI"},
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
# Pre-fix this assertion fails — the handler swallows the
|
||||
# KeyError raised by the chat-completions transformer and falls
|
||||
# back to the passthrough_chat_handler which yields a different
|
||||
# response_cost value. Post-fix, the Responses transformer
|
||||
# succeeds and we get the mocked 0.000050.
|
||||
assert result is not None
|
||||
assert result["kwargs"]["response_cost"] == 0.000050
|
||||
assert result["kwargs"]["model"] == "gpt-4o"
|
||||
|
||||
# `completion_cost` must be called with the responses call type
|
||||
# and a `ResponsesAPIResponse` (not a `ModelResponse`).
|
||||
mock_completion_cost.assert_called_once()
|
||||
call_kwargs = mock_completion_cost.call_args[1]
|
||||
assert call_kwargs["call_type"] == "responses"
|
||||
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
assert isinstance(call_kwargs["completion_response"], ResponsesAPIResponse), (
|
||||
"completion_response must be a ResponsesAPIResponse; passing a "
|
||||
"chat-completions ModelResponse means the Responses transformer "
|
||||
"isn't being used and we're back in the bug."
|
||||
)
|
||||
|
||||
|
||||
class TestOpenAIPassthroughIntegration:
|
||||
"""Integration tests for OpenAI passthrough cost tracking"""
|
||||
|
|
@ -872,6 +982,126 @@ class TestOpenAIPassthroughIntegration:
|
|||
)
|
||||
assert self.handler.is_openai_route("") == False
|
||||
|
||||
def test_is_supported_openai_endpoint_includes_responses_api(self):
|
||||
"""Regression test for the outer dispatch gate.
|
||||
|
||||
`_is_supported_openai_endpoint` is the gate that decides whether the
|
||||
OpenAI handler runs for a given URL. Before this gate accepted the
|
||||
Responses API, calls to `/v1/responses` would fail the gate and the
|
||||
handler's `elif is_responses:` branch was unreachable in the live
|
||||
success-handler pipeline — every Responses-API call landed in
|
||||
`LiteLLM_SpendLogs` with zero tokens / zero spend even though the
|
||||
handler had a Responses branch internally.
|
||||
|
||||
This test exercises the dispatch decision directly so future
|
||||
refactors of `_is_supported_openai_endpoint` can't silently
|
||||
remove Responses from the OR-chain without a test failure.
|
||||
"""
|
||||
# Responses must be supported on api.openai.com and openai.azure.com.
|
||||
assert (
|
||||
self.handler._is_supported_openai_endpoint(
|
||||
"https://api.openai.com/v1/responses"
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert (
|
||||
self.handler._is_supported_openai_endpoint(
|
||||
"https://openai.azure.com/v1/responses"
|
||||
)
|
||||
is True
|
||||
)
|
||||
# The other supported endpoints stay supported (no regression).
|
||||
assert (
|
||||
self.handler._is_supported_openai_endpoint(
|
||||
"https://api.openai.com/v1/chat/completions"
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert (
|
||||
self.handler._is_supported_openai_endpoint(
|
||||
"https://api.openai.com/v1/images/generations"
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert (
|
||||
self.handler._is_supported_openai_endpoint(
|
||||
"https://api.openai.com/v1/images/edits"
|
||||
)
|
||||
is True
|
||||
)
|
||||
# Unsupported OpenAI endpoints (e.g. /v1/models) still return False.
|
||||
assert (
|
||||
self.handler._is_supported_openai_endpoint(
|
||||
"https://api.openai.com/v1/models"
|
||||
)
|
||||
is False
|
||||
)
|
||||
|
||||
@patch(
|
||||
"litellm.proxy.pass_through_endpoints.llm_provider_handlers.openai_passthrough_logging_handler.OpenAIPassthroughLoggingHandler.openai_passthrough_handler"
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_success_handler_dispatches_responses_api_to_openai_handler(
|
||||
self, mock_openai_handler
|
||||
):
|
||||
"""End-to-end dispatch test for the Responses API path.
|
||||
|
||||
Pre-fix: `_is_supported_openai_endpoint` returned False for
|
||||
`/v1/responses` URLs, so the OpenAI handler was never called.
|
||||
This test would fail (mock never invoked) on the un-fixed
|
||||
success_handler — passes only when the dispatch gate accepts
|
||||
Responses URLs.
|
||||
"""
|
||||
mock_openai_handler.return_value = {
|
||||
"result": {"id": "resp_abc123"},
|
||||
"kwargs": {
|
||||
"response_cost": 0.0001,
|
||||
"model": "gpt-4o",
|
||||
"custom_llm_provider": "openai",
|
||||
},
|
||||
}
|
||||
|
||||
mock_httpx_response = MagicMock(spec=httpx.Response)
|
||||
mock_httpx_response.text = (
|
||||
'{"id": "resp_abc123", "object": "response", '
|
||||
'"output": [], "usage": {"input_tokens": 5, "output_tokens": 3}}'
|
||||
)
|
||||
|
||||
mock_logging_obj = AsyncMock()
|
||||
mock_logging_obj.model_call_details = {}
|
||||
mock_logging_obj.async_success_handler = AsyncMock()
|
||||
|
||||
passthrough_payload = PassthroughStandardLoggingPayload(
|
||||
url="https://api.openai.com/v1/responses",
|
||||
request_body={"model": "gpt-4o", "input": "Hello"},
|
||||
request_method="POST",
|
||||
)
|
||||
|
||||
await self.handler.pass_through_async_success_handler(
|
||||
httpx_response=mock_httpx_response,
|
||||
response_body={
|
||||
"id": "resp_abc123",
|
||||
"object": "response",
|
||||
"output": [],
|
||||
"usage": {"input_tokens": 5, "output_tokens": 3},
|
||||
},
|
||||
logging_obj=mock_logging_obj,
|
||||
url_route="https://api.openai.com/v1/responses",
|
||||
result="",
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
cache_hit=False,
|
||||
request_body={"model": "gpt-4o", "input": "Hello"},
|
||||
passthrough_logging_payload=passthrough_payload,
|
||||
)
|
||||
|
||||
# The OpenAI handler MUST have been invoked. Pre-fix the dispatch
|
||||
# gate filtered Responses URLs out and the mock was never called.
|
||||
mock_openai_handler.assert_called_once()
|
||||
# And we can verify it was dispatched with the Responses URL.
|
||||
call_kwargs = mock_openai_handler.call_args.kwargs
|
||||
assert call_kwargs["url_route"] == "https://api.openai.com/v1/responses"
|
||||
|
||||
@patch(
|
||||
"litellm.proxy.pass_through_endpoints.llm_provider_handlers.openai_passthrough_logging_handler.OpenAIPassthroughLoggingHandler.openai_passthrough_handler"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue