mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(complexity_router): log the classifier request on chat completions too
The classifier read its metadata only from litellm_metadata, which the proxy populates just for LITELLM_METADATA_ROUTES (/v1/messages, /v1/responses, ...); /v1/chat/completions puts it under metadata, so the classifier call arrived unattributed and _should_track_cost_callback dropped it, leaving no spend-log row at all for the captured request body to show up in. Also log response_format in the wire shape litellm actually sends (type_to_response_format_param) instead of the bare pydantic JSON schema Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
2a84c39762
commit
78207064d8
2 changed files with 29 additions and 2 deletions
|
|
@ -25,6 +25,7 @@ from pydantic import BaseModel
|
|||
from litellm._logging import verbose_router_logger
|
||||
from litellm.constants import RETURN_RAW_MODEL_NAME_METADATA_KEY
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.llms.base_llm.base_utils import type_to_response_format_param
|
||||
from litellm.types.utils import ModelResponse
|
||||
|
||||
from .config import (
|
||||
|
|
@ -427,13 +428,14 @@ class ComplexityRouter(CustomLogger):
|
|||
# attributed to the calling key/team instead of being dropped. Excludes the
|
||||
# parent request's budget reservation, which the routed completion (not this
|
||||
# internal classifier call) is responsible for reconciling.
|
||||
metadata = _classifier_call_metadata((request_kwargs or {}).get("litellm_metadata"))
|
||||
request_metadata = (request_kwargs or {}).get("litellm_metadata") or (request_kwargs or {}).get("metadata")
|
||||
metadata = _classifier_call_metadata(request_metadata)
|
||||
|
||||
proxy_server_request = {
|
||||
"body": {
|
||||
"model": llm_config.model,
|
||||
"messages": [{"role": "user", "content": classification_prompt}],
|
||||
"response_format": TierClassification.model_json_schema(),
|
||||
"response_format": type_to_response_format_param(TierClassification),
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -1417,6 +1417,24 @@ class TestLLMClassifier:
|
|||
call_kwargs = mock_router_instance.acompletion.call_args.kwargs
|
||||
assert call_kwargs["metadata"] == request_metadata
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_aclassify_forwards_metadata_key_used_by_chat_completions(
|
||||
self, llm_complexity_router, mock_router_instance
|
||||
):
|
||||
"""/v1/chat/completions puts the request metadata under "metadata", not "litellm_metadata".
|
||||
|
||||
Only the routes in LITELLM_METADATA_ROUTES (/v1/messages, /v1/responses, ...) get a
|
||||
"litellm_metadata" bucket; chat completions gets "metadata". Reading only
|
||||
"litellm_metadata" leaves the classifier call unattributed on the most common route,
|
||||
so _should_track_cost_callback drops it and no spend-log row is written at all,
|
||||
which also makes the captured request body unreachable in the Logs UI.
|
||||
"""
|
||||
mock_router_instance.acompletion = AsyncMock(return_value=_llm_response('{"tier": "SIMPLE"}'))
|
||||
request_metadata = {"user_api_key": "sk-abc", "user_api_key_team_id": "team-1"}
|
||||
await llm_complexity_router.aclassify("hi", request_kwargs={"metadata": request_metadata})
|
||||
call_kwargs = mock_router_instance.acompletion.call_args.kwargs
|
||||
assert call_kwargs["metadata"] == request_metadata
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_aclassify_captures_request_body_in_proxy_server_request(
|
||||
self, llm_complexity_router, mock_router_instance
|
||||
|
|
@ -1439,6 +1457,13 @@ class TestLLMClassifier:
|
|||
assert body["model"] == "haiku-classifier"
|
||||
assert body["messages"] == call_kwargs["messages"]
|
||||
assert "explain quantum tunneling in depth" in body["messages"][0]["content"]
|
||||
assert body["response_format"]["type"] == "json_schema"
|
||||
assert body["response_format"]["json_schema"]["schema"]["properties"]["tier"]["enum"] == [
|
||||
"SIMPLE",
|
||||
"MEDIUM",
|
||||
"COMPLEX",
|
||||
"REASONING",
|
||||
]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_aclassify_strips_budget_reservation_from_classifier_metadata(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue