From 78207064d8d426b2e408c6572f780ccda613eaf6 Mon Sep 17 00:00:00 2001 From: tin Date: Thu, 30 Jul 2026 01:09:37 +0000 Subject: [PATCH] fix(complexity_router): log the classifier request on chat completions too The classifier read its metadata only from litellm_metadata, which the proxy populates just for LITELLM_METADATA_ROUTES (/v1/messages, /v1/responses, ...); /v1/chat/completions puts it under metadata, so the classifier call arrived unattributed and _should_track_cost_callback dropped it, leaving no spend-log row at all for the captured request body to show up in. Also log response_format in the wire shape litellm actually sends (type_to_response_format_param) instead of the bare pydantic JSON schema Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../complexity_router/complexity_router.py | 6 +++-- .../router_strategy/test_complexity_router.py | 25 +++++++++++++++++++ 2 files changed, 29 insertions(+), 2 deletions(-) diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index 836c2e9d4c0..4bc847c923e 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -25,6 +25,7 @@ from pydantic import BaseModel from litellm._logging import verbose_router_logger from litellm.constants import RETURN_RAW_MODEL_NAME_METADATA_KEY from litellm.integrations.custom_logger import CustomLogger +from litellm.llms.base_llm.base_utils import type_to_response_format_param from litellm.types.utils import ModelResponse from .config import ( @@ -427,13 +428,14 @@ class ComplexityRouter(CustomLogger): # attributed to the calling key/team instead of being dropped. Excludes the # parent request's budget reservation, which the routed completion (not this # internal classifier call) is responsible for reconciling. - metadata = _classifier_call_metadata((request_kwargs or {}).get("litellm_metadata")) + request_metadata = (request_kwargs or {}).get("litellm_metadata") or (request_kwargs or {}).get("metadata") + metadata = _classifier_call_metadata(request_metadata) proxy_server_request = { "body": { "model": llm_config.model, "messages": [{"role": "user", "content": classification_prompt}], - "response_format": TierClassification.model_json_schema(), + "response_format": type_to_response_format_param(TierClassification), } } diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index a9b86c16c33..1feeb150c87 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -1417,6 +1417,24 @@ class TestLLMClassifier: call_kwargs = mock_router_instance.acompletion.call_args.kwargs assert call_kwargs["metadata"] == request_metadata + @pytest.mark.asyncio + async def test_aclassify_forwards_metadata_key_used_by_chat_completions( + self, llm_complexity_router, mock_router_instance + ): + """/v1/chat/completions puts the request metadata under "metadata", not "litellm_metadata". + + Only the routes in LITELLM_METADATA_ROUTES (/v1/messages, /v1/responses, ...) get a + "litellm_metadata" bucket; chat completions gets "metadata". Reading only + "litellm_metadata" leaves the classifier call unattributed on the most common route, + so _should_track_cost_callback drops it and no spend-log row is written at all, + which also makes the captured request body unreachable in the Logs UI. + """ + mock_router_instance.acompletion = AsyncMock(return_value=_llm_response('{"tier": "SIMPLE"}')) + request_metadata = {"user_api_key": "sk-abc", "user_api_key_team_id": "team-1"} + await llm_complexity_router.aclassify("hi", request_kwargs={"metadata": request_metadata}) + call_kwargs = mock_router_instance.acompletion.call_args.kwargs + assert call_kwargs["metadata"] == request_metadata + @pytest.mark.asyncio async def test_aclassify_captures_request_body_in_proxy_server_request( self, llm_complexity_router, mock_router_instance @@ -1439,6 +1457,13 @@ class TestLLMClassifier: assert body["model"] == "haiku-classifier" assert body["messages"] == call_kwargs["messages"] assert "explain quantum tunneling in depth" in body["messages"][0]["content"] + assert body["response_format"]["type"] == "json_schema" + assert body["response_format"]["json_schema"]["schema"]["properties"]["tier"]["enum"] == [ + "SIMPLE", + "MEDIUM", + "COMPLEX", + "REASONING", + ] @pytest.mark.asyncio async def test_aclassify_strips_budget_reservation_from_classifier_metadata(