This commit is contained in:
samarth9008 2026-04-03 13:41:21 -04:00
parent d4a3a5e530
commit ec60ecaca5
2 changed files with 103 additions and 11 deletions

View file

@ -2598,10 +2598,12 @@ class BaseLLMHTTPHandler:
provider_config=responses_api_provider_config,
)
return responses_api_provider_config.transform_get_response_api_response(
result = responses_api_provider_config.transform_get_response_api_response(
raw_response=response,
logging_obj=logging_obj,
)
result._hidden_params["response_cost"] = 0.0
return result
async def async_get_responses(
self,
@ -2673,10 +2675,12 @@ class BaseLLMHTTPHandler:
provider_config=responses_api_provider_config,
)
return responses_api_provider_config.transform_get_response_api_response(
result = responses_api_provider_config.transform_get_response_api_response(
raw_response=response,
logging_obj=logging_obj,
)
result._hidden_params["response_cost"] = 0.0
return result
#####################################################################
################ LIST RESPONSES INPUT ITEMS HANDLER ###########################
@ -4495,9 +4499,9 @@ class BaseLLMHTTPHandler:
# Second: Execute agentic loop
# Add custom_llm_provider to kwargs so the agentic loop can reconstruct the full model name
kwargs_with_provider = kwargs.copy() if kwargs else {}
kwargs_with_provider[
"custom_llm_provider"
] = custom_llm_provider
kwargs_with_provider["custom_llm_provider"] = (
custom_llm_provider
)
agentic_response = await callback.async_run_agentic_loop(
tools=tool_calls,
model=model,
@ -4613,9 +4617,9 @@ class BaseLLMHTTPHandler:
# Second: Execute agentic loop
# Add custom_llm_provider to kwargs so the agentic loop can reconstruct the full model name
kwargs_with_provider = kwargs.copy() if kwargs else {}
kwargs_with_provider[
"custom_llm_provider"
] = custom_llm_provider
kwargs_with_provider["custom_llm_provider"] = (
custom_llm_provider
)
agentic_response = (
await callback.async_run_chat_completion_agentic_loop(
tools=tool_calls,
@ -5099,7 +5103,10 @@ class BaseLLMHTTPHandler:
_is_async: bool = False,
fake_stream: bool = False,
litellm_metadata: Optional[Dict[str, Any]] = None,
) -> Union[ImageResponse, Coroutine[Any, Any, ImageResponse],]:
) -> Union[
ImageResponse,
Coroutine[Any, Any, ImageResponse],
]:
"""
Handles image edit requests.
@ -5311,7 +5318,10 @@ class BaseLLMHTTPHandler:
fake_stream: bool = False,
litellm_metadata: Optional[Dict[str, Any]] = None,
api_key: Optional[str] = None,
) -> Union[ImageResponse, Coroutine[Any, Any, ImageResponse],]:
) -> Union[
ImageResponse,
Coroutine[Any, Any, ImageResponse],
]:
"""
Handles image generation requests.
When _is_async=True, returns a coroutine instead of making the call directly.
@ -5551,7 +5561,10 @@ class BaseLLMHTTPHandler:
fake_stream: bool = False,
litellm_metadata: Optional[Dict[str, Any]] = None,
api_key: Optional[str] = None,
) -> Union[VideoObject, Coroutine[Any, Any, VideoObject],]:
) -> Union[
VideoObject,
Coroutine[Any, Any, VideoObject],
]:
"""
Handles video generation requests.
When _is_async=True, returns a coroutine instead of making the call directly.

View file

@ -1282,3 +1282,82 @@ class TestPhaseParameter:
assert validated[0]["phase"] == "commentary"
assert validated[1]["phase"] == "final_answer"
assert "phase" not in validated[2]
class TestGetResponsesZeroCost:
"""Test that GET /v1/responses/{response_id} does not incur duplicate costs (bug #25015)."""
def _make_fake_response(self):
"""Create a fake httpx response with usage data (as OpenAI would return)."""
fake_response_json = {
"id": "resp_abc123",
"object": "response",
"created_at": 1700000000,
"status": "completed",
"model": "o3-deep-research",
"output": [],
"usage": {
"input_tokens": 100,
"output_tokens": 50,
"total_tokens": 150,
},
}
return httpx.Response(
status_code=200,
json=fake_response_json,
request=httpx.Request("GET", "https://api.openai.com/v1/responses/resp_abc123"),
)
def test_get_responses_sets_zero_cost(self):
"""
Verify that get_responses (sync) sets response_cost=0.0 in _hidden_params,
so the cost pipeline does not re-bill for a retrieval request.
"""
from litellm.llms.custom_httpx.http_handler import HTTPHandler
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
handler = BaseLLMHTTPHandler()
config = OpenAIResponsesAPIConfig()
logging_obj = MagicMock()
litellm_params = GenericLiteLLMParams(api_base="https://api.openai.com/v1/responses")
mock_client = MagicMock(spec=HTTPHandler)
mock_client.get.return_value = self._make_fake_response()
result = handler.get_responses(
response_id="resp_abc123",
responses_api_provider_config=config,
litellm_params=litellm_params,
logging_obj=logging_obj,
client=mock_client,
)
assert result._hidden_params["response_cost"] == 0.0
@pytest.mark.asyncio
async def test_async_get_responses_sets_zero_cost(self):
"""
Verify that async_get_responses sets response_cost=0.0 in _hidden_params,
so the cost pipeline does not re-bill for a retrieval request.
"""
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
handler = BaseLLMHTTPHandler()
config = OpenAIResponsesAPIConfig()
logging_obj = MagicMock()
litellm_params = GenericLiteLLMParams(api_base="https://api.openai.com/v1/responses")
mock_client = AsyncMock(spec=AsyncHTTPHandler)
mock_client.get.return_value = self._make_fake_response()
result = await handler.async_get_responses(
response_id="resp_abc123",
responses_api_provider_config=config,
litellm_params=litellm_params,
logging_obj=logging_obj,
client=mock_client,
custom_llm_provider="openai",
)
assert result._hidden_params["response_cost"] == 0.0