mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix: skip overhead metric for chat completions route types
Chat completions already set litellm_overhead_time_ms in the SDK layer. Skip our overhead calculation for acompletion/completion routes to avoid interfering with object responses.
This commit is contained in:
parent
084432f9ed
commit
b69a4a47ac
1 changed files with 7 additions and 2 deletions
|
|
@ -1427,9 +1427,14 @@ class ProxyBaseLLMRequestProcessing:
|
|||
|
||||
response = responses[1]
|
||||
|
||||
# GH#30566: overhead for non-chat-completions routes
|
||||
# GH#30566: overhead for non-chat-completions routes.
|
||||
# Skip chat routes (acompletion/completion) because the SDK already
|
||||
# sets litellm_overhead_time_ms in litellm.utils.completion().
|
||||
_hidden_params = getattr(response, "_hidden_params", {}) or {}
|
||||
if not _hidden_params.get("litellm_overhead_time_ms"):
|
||||
if (
|
||||
not _hidden_params.get("litellm_overhead_time_ms")
|
||||
and route_type not in ("acompletion", "completion")
|
||||
):
|
||||
end_time = datetime.now()
|
||||
_logging_obj = self.data.get("litellm_logging_obj")
|
||||
if _logging_obj is not None:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue