fix(complexity_router): log the classifier request on chat completions too

The classifier read its metadata only from litellm_metadata, which the proxy
populates just for LITELLM_METADATA_ROUTES (/v1/messages, /v1/responses, ...);
/v1/chat/completions puts it under metadata, so the classifier call arrived
unattributed and _should_track_cost_callback dropped it, leaving no spend-log
row at all for the captured request body to show up in.

Also log response_format in the wire shape litellm actually sends
(type_to_response_format_param) instead of the bare pydantic JSON schema

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
tin 2026-07-30 01:09:37 +00:00 • committed by Tin Chi Lo
parent 2a84c39762
commit 78207064d8
2 changed files with 29 additions and 2 deletions

View file

@ -25,6 +25,7 @@ from pydantic import BaseModel
from litellm._logging import verbose_router_logger
from litellm.constants import RETURN_RAW_MODEL_NAME_METADATA_KEY
from litellm.integrations.custom_logger import CustomLogger
from litellm.llms.base_llm.base_utils import type_to_response_format_param
from litellm.types.utils import ModelResponse
from .config import (
@ -427,13 +428,14 @@ class ComplexityRouter(CustomLogger):
# attributed to the calling key/team instead of being dropped. Excludes the
# parent request's budget reservation, which the routed completion (not this
# internal classifier call) is responsible for reconciling.
metadata = _classifier_call_metadata((request_kwargs or {}).get("litellm_metadata"))
request_metadata = (request_kwargs or {}).get("litellm_metadata") or (request_kwargs or {}).get("metadata")
metadata = _classifier_call_metadata(request_metadata)
proxy_server_request = {
"body": {
"model": llm_config.model,
"messages": [{"role": "user", "content": classification_prompt}],
"response_format": TierClassification.model_json_schema(),
"response_format": type_to_response_format_param(TierClassification),
}
}

View file

@ -1417,6 +1417,24 @@ class TestLLMClassifier:
call_kwargs = mock_router_instance.acompletion.call_args.kwargs
assert call_kwargs["metadata"] == request_metadata
@pytest.mark.asyncio
async def test_aclassify_forwards_metadata_key_used_by_chat_completions(
self, llm_complexity_router, mock_router_instance
):
"""/v1/chat/completions puts the request metadata under "metadata", not "litellm_metadata".
Only the routes in LITELLM_METADATA_ROUTES (/v1/messages, /v1/responses, ...) get a
"litellm_metadata" bucket; chat completions gets "metadata". Reading only
"litellm_metadata" leaves the classifier call unattributed on the most common route,
so _should_track_cost_callback drops it and no spend-log row is written at all,
which also makes the captured request body unreachable in the Logs UI.
"""
mock_router_instance.acompletion = AsyncMock(return_value=_llm_response('{"tier": "SIMPLE"}'))
request_metadata = {"user_api_key": "sk-abc", "user_api_key_team_id": "team-1"}
await llm_complexity_router.aclassify("hi", request_kwargs={"metadata": request_metadata})
call_kwargs = mock_router_instance.acompletion.call_args.kwargs
assert call_kwargs["metadata"] == request_metadata
@pytest.mark.asyncio
async def test_aclassify_captures_request_body_in_proxy_server_request(
self, llm_complexity_router, mock_router_instance
@ -1439,6 +1457,13 @@ class TestLLMClassifier:
assert body["model"] == "haiku-classifier"
assert body["messages"] == call_kwargs["messages"]
assert "explain quantum tunneling in depth" in body["messages"][0]["content"]
assert body["response_format"]["type"] == "json_schema"
assert body["response_format"]["json_schema"]["schema"]["properties"]["tier"]["enum"] == [
"SIMPLE",
"MEDIUM",
"COMPLEX",
"REASONING",
]
@pytest.mark.asyncio
async def test_aclassify_strips_budget_reservation_from_classifier_metadata(