From b69a4a47ac8558792b228e0e9de7047d60173a6e Mon Sep 17 00:00:00 2001 From: factnn <166481866+factnn@users.noreply.github.com> Date: Thu, 2 Jul 2026 22:01:33 +0800 Subject: [PATCH] fix: skip overhead metric for chat completions route types Chat completions already set litellm_overhead_time_ms in the SDK layer. Skip our overhead calculation for acompletion/completion routes to avoid interfering with object responses. --- litellm/proxy/common_request_processing.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 4349ec130c4..cb93c98cfae 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -1427,9 +1427,14 @@ class ProxyBaseLLMRequestProcessing: response = responses[1] - # GH#30566: overhead for non-chat-completions routes + # GH#30566: overhead for non-chat-completions routes. + # Skip chat routes (acompletion/completion) because the SDK already + # sets litellm_overhead_time_ms in litellm.utils.completion(). _hidden_params = getattr(response, "_hidden_params", {}) or {} - if not _hidden_params.get("litellm_overhead_time_ms"): + if ( + not _hidden_params.get("litellm_overhead_time_ms") + and route_type not in ("acompletion", "completion") + ): end_time = datetime.now() _logging_obj = self.data.get("litellm_logging_obj") if _logging_obj is not None: