fix: skip overhead metric for chat completions route types

Chat completions already set litellm_overhead_time_ms in the SDK
layer. Skip our overhead calculation for acompletion/completion
routes to avoid interfering with object responses.
This commit is contained in:
factnn 2026-07-02 22:01:33 +08:00
parent 084432f9ed
commit b69a4a47ac

View file

@ -1427,9 +1427,14 @@ class ProxyBaseLLMRequestProcessing:
response = responses[1]
# GH#30566: overhead for non-chat-completions routes
# GH#30566: overhead for non-chat-completions routes.
# Skip chat routes (acompletion/completion) because the SDK already
# sets litellm_overhead_time_ms in litellm.utils.completion().
_hidden_params = getattr(response, "_hidden_params", {}) or {}
if not _hidden_params.get("litellm_overhead_time_ms"):
if (
not _hidden_params.get("litellm_overhead_time_ms")
and route_type not in ("acompletion", "completion")
):
end_time = datetime.now()
_logging_obj = self.data.get("litellm_logging_obj")
if _logging_obj is not None: