diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 4349ec130c4..cb93c98cfae 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -1427,9 +1427,14 @@ class ProxyBaseLLMRequestProcessing: response = responses[1] - # GH#30566: overhead for non-chat-completions routes + # GH#30566: overhead for non-chat-completions routes. + # Skip chat routes (acompletion/completion) because the SDK already + # sets litellm_overhead_time_ms in litellm.utils.completion(). _hidden_params = getattr(response, "_hidden_params", {}) or {} - if not _hidden_params.get("litellm_overhead_time_ms"): + if ( + not _hidden_params.get("litellm_overhead_time_ms") + and route_type not in ("acompletion", "completion") + ): end_time = datetime.now() _logging_obj = self.data.get("litellm_logging_obj") if _logging_obj is not None: