mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-10 22:41:41 +00:00
Merge 75b049d3cc into 7083c47998
This commit is contained in:
commit
fc60ae1fe8
2 changed files with 70 additions and 0 deletions
|
|
@ -2293,6 +2293,30 @@ class ProxyBaseLLMRequestProcessing:
|
|||
|
||||
response = responses[1]
|
||||
|
||||
# GH#30566: set overhead duration for non-chat-completions
|
||||
# routes (/v1/messages, /v1/responses). Chat completions
|
||||
# already have litellm_overhead_time_ms from the SDK.
|
||||
# Only set the overhead field directly (don't call
|
||||
# update_response_metadata which also touches cost).
|
||||
_overhead_hidden_params = getattr(response, "_hidden_params", {}) or {}
|
||||
if not _overhead_hidden_params.get("litellm_overhead_time_ms") and route_type not in (
|
||||
"acompletion",
|
||||
"completion",
|
||||
):
|
||||
end_time = datetime.now()
|
||||
_logging_obj = self.data.get("litellm_logging_obj")
|
||||
if _logging_obj is not None and _logging_obj.start_time is not None:
|
||||
overhead_ms = (end_time - _logging_obj.start_time).total_seconds() * 1000 - (
|
||||
_logging_obj.model_call_details.get("llm_api_duration_ms", 0)
|
||||
)
|
||||
if not isinstance(_overhead_hidden_params, dict):
|
||||
_overhead_hidden_params = {}
|
||||
_overhead_hidden_params["litellm_overhead_time_ms"] = overhead_ms
|
||||
if hasattr(response, "_hidden_params"):
|
||||
response._hidden_params = _overhead_hidden_params
|
||||
elif isinstance(response, dict):
|
||||
response["_hidden_params"] = _overhead_hidden_params
|
||||
|
||||
_exception_raised = False
|
||||
try:
|
||||
hidden_params = get_hidden_params_dict(response)
|
||||
|
|
|
|||
|
|
@ -8,6 +8,8 @@ through _hidden_params to the x-litellm-callback-duration-ms response header.
|
|||
import datetime
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm.litellm_core_utils.llm_response_utils.response_metadata as response_metadata_mod
|
||||
import litellm.proxy.common_request_processing as common_request_processing_mod
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
|
|
@ -91,6 +93,50 @@ class TestCallbackDurationMs:
|
|||
# overhead should also be set
|
||||
assert hidden.get("litellm_overhead_time_ms") is not None
|
||||
|
||||
def test_overhead_computed_for_routes_without_pre_existing_value(self):
|
||||
"""GH#30566: overhead is set even when _hidden_params
|
||||
does not already contain litellm_overhead_time_ms.
|
||||
This simulates non-chat-completions routes that skip
|
||||
the SDK-level update_response_metadata call."""
|
||||
result = ModelResponse()
|
||||
logging_obj = self._make_logging_obj(llm_api_duration_ms=900.0)
|
||||
logging_obj._response_cost_calculator = MagicMock(return_value=0.001)
|
||||
logging_obj.litellm_call_id = "test-gh30566"
|
||||
|
||||
start = datetime.datetime(2025, 1, 1, 0, 0, 0)
|
||||
end = datetime.datetime(2025, 1, 1, 0, 0, 1)
|
||||
|
||||
update_response_metadata(
|
||||
result=result,
|
||||
logging_obj=logging_obj,
|
||||
model="gpt-4",
|
||||
kwargs={},
|
||||
start_time=start,
|
||||
end_time=end,
|
||||
)
|
||||
|
||||
hidden = result._hidden_params
|
||||
assert hidden.get("litellm_overhead_time_ms") == pytest.approx(100.0, rel=0.01)
|
||||
assert hidden.get("_response_ms") == pytest.approx(1000.0, rel=0.01)
|
||||
|
||||
def test_overhead_guard_skips_when_already_present(self):
|
||||
"""GH#30566: the guard in base_process_llm_request prevents
|
||||
calling update_response_metadata when litellm_overhead_time_ms
|
||||
is already set by the SDK layer (e.g. /v1/chat/completions).
|
||||
This test verifies the guard condition directly."""
|
||||
# Chat-completions path: overhead already populated
|
||||
result = ModelResponse()
|
||||
result._hidden_params = {"litellm_overhead_time_ms": 50.0}
|
||||
hidden_params = getattr(result, "_hidden_params", {}) or {}
|
||||
should_skip = bool(hidden_params.get("litellm_overhead_time_ms"))
|
||||
assert should_skip is True
|
||||
|
||||
# Non-chat path (/v1/messages, /v1/responses): no overhead yet
|
||||
result2 = ModelResponse()
|
||||
hidden_params2 = getattr(result2, "_hidden_params", {}) or {}
|
||||
should_skip2 = bool(hidden_params2.get("litellm_overhead_time_ms"))
|
||||
assert should_skip2 is False
|
||||
|
||||
|
||||
class TestDictResultsSkipMetadataUpdate:
|
||||
"""Regression for /v1/messages cost-breakdown clobbering: AnthropicMessagesResponse
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue