fix: set overhead duration metric for all route types

update_response_metadata was only called in litellm.utils.completion()
for /v1/chat/completions. Routes like /v1/messages (Anthropic) and
/v1/responses (Responses API) skipped this call.

Call update_response_metadata from base_process_llm_request after
the LLM response is received, but only when litellm_overhead_time_ms
is not already set, so chat completions keep their SDK-level timing.

Fixes #30566
This commit is contained in:
factnn 2026-06-17 15:04:07 +08:00
parent 79a6b8f7f0
commit 54e95af071

View file

@ -1366,6 +1366,8 @@ class ProxyBaseLLMRequestProcessing:
llm_router=llm_router,
)
self.data["start_time"] = datetime.now()
# Defer async logging when post-call guardrails are configured so the
# StandardLoggingPayload is built after guardrails write to metadata.
# Cache the result to avoid scanning litellm.callbacks twice.
@ -1427,6 +1429,23 @@ class ProxyBaseLLMRequestProcessing:
response = responses[1]
# GH#30566: overhead for non-chat-completions routes
_hidden_params = getattr(response, "_hidden_params", {}) or {}
if not _hidden_params.get("litellm_overhead_time_ms"):
end_time = datetime.now()
from litellm.litellm_core_utils.llm_response_utils.response_metadata import (
update_response_metadata,
)
update_response_metadata(
result=response,
logging_obj=self.data.get("litellm_logging_obj"),
model=self.data.get("model"),
kwargs=self.data,
start_time=self.data.get("start_time", end_time),
end_time=end_time,
)
_exception_raised = False
try:
hidden_params = getattr(response, "_hidden_params", {}) or {}