From 779441b47ba564696c35e86a66527e683c1fad31 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 12 Aug 2026 00:30:19 +0000 Subject: [PATCH 01/31] fix(bedrock_mantle): source per-request AWS credential params from litellm_params when signing chat completions Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/custom_httpx/llm_http_handler.py | 26 ++++++- .../test_bedrock_mantle_transformation.py | 77 +++++++++++++++++++ .../custom_httpx/test_llm_http_handler.py | 19 +++++ 3 files changed, 120 insertions(+), 2 deletions(-) diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 721b9545ac1..dfc234d08e0 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -5,7 +5,7 @@ import ssl from collections.abc import AsyncIterator, Coroutine, Iterator, Mapping from contextlib import asynccontextmanager from functools import lru_cache -from types import ModuleType +from types import MappingProxyType, ModuleType from typing import TYPE_CHECKING, Any, Final, Literal, Optional, TypedDict, TypeVar, Union, cast, get_type_hints from urllib.parse import parse_qs, urlencode, urlparse, urlunparse @@ -20,6 +20,7 @@ from litellm._logging import _redact_string, verbose_logger from litellm.anthropic_beta_headers_manager import update_headers_with_filtered_beta from litellm.constants import REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES from litellm.litellm_core_utils.asyncify import run_async_function +from litellm.litellm_core_utils.get_litellm_params import AWS_CREDENTIAL_KWARGS_KEYS from litellm.litellm_core_utils.realtime_streaming import RealTimeStreaming from litellm.litellm_core_utils.url_utils import encode_url_path_segment from litellm.llms.base_llm.anthropic_messages.transformation import ( @@ -252,6 +253,24 @@ def _has_pre_call_deployment_hook(logging_obj: LiteLLMLoggingObj) -> bool: return False +def _aws_signing_overrides( + optional_params: Mapping[str, Any], litellm_params: Mapping[str, Any] +) -> Mapping[str, Any]: + """AWS credential params for SigV4 signers that read them off optional_params. + + Only `bedrock`/`sagemaker` keep `aws_*` in optional_params: every other provider + spreads optional_params into the request body, so the params are stripped there + and survive on litellm_params alone. + """ + return MappingProxyType( + { + key: litellm_params[key] + for key in AWS_CREDENTIAL_KWARGS_KEYS + if optional_params.get(key) is None and litellm_params.get(key) is not None + } + ) + + class BaseLLMHTTPHandler: async def _make_common_async_call( self, @@ -495,7 +514,10 @@ class BaseLLMHTTPHandler: headers, signed_json_body = provider_config.sign_request( headers=headers, - optional_params=optional_params, + optional_params={ + **optional_params, + **_aws_signing_overrides(optional_params, litellm_params), + }, request_data=data, api_base=api_base, api_key=api_key, diff --git a/tests/test_litellm/llms/bedrock_mantle/test_bedrock_mantle_transformation.py b/tests/test_litellm/llms/bedrock_mantle/test_bedrock_mantle_transformation.py index 275fb460b9f..468cc9b9130 100644 --- a/tests/test_litellm/llms/bedrock_mantle/test_bedrock_mantle_transformation.py +++ b/tests/test_litellm/llms/bedrock_mantle/test_bedrock_mantle_transformation.py @@ -489,6 +489,83 @@ class TestBedrockMantleChatAuth: assert "/us-east-2/bedrock/aws4_request" in authorization assert requests[0]["url"].startswith("https://bedrock-mantle.us-east-2.api.aws") + def test_completion_per_request_role_reaches_signer_and_not_the_body( + self, monkeypatch + ): + # Per-request aws_role_name/aws_session_name are stripped from optional_params + # for non-bedrock providers, so they must be sourced from litellm_params at + # signing time, and must never be serialized into the provider request body. + from botocore.credentials import Credentials + + from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM + + for var in ( + "BEDROCK_MANTLE_API_KEY", + "AWS_BEARER_TOKEN_BEDROCK", + "BEDROCK_MANTLE_API_BASE", + ): + monkeypatch.delenv(var, raising=False) + + credential_calls = [] + + def fake_get_credentials(self, **kwargs): + credential_calls.append(kwargs) + return Credentials( + access_key="ASIAEXAMPLE", + secret_key="YXNzdW1lZC1yb2xlLXNlY3JldC1hc3N1bWVk", + token="assumed-session-token", + ) + + monkeypatch.setattr(BaseAWSLLM, "get_credentials", fake_get_credentials) + + requests = [] + + def mock_post(self, url, data=None, headers=None, **kwargs): + raw_body = data.decode("utf-8") if isinstance(data, bytes) else data + requests.append({"headers": headers or {}, "body": json.loads(raw_body or "{}")}) + return httpx.Response( + status_code=200, + json={ + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1733529600, + "model": "google.gemma-4-31b", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok"}, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": 1, + "completion_tokens": 1, + "total_tokens": 2, + }, + }, + request=httpx.Request("POST", url), + ) + + with patch( + "litellm.llms.custom_httpx.http_handler.HTTPHandler.post", mock_post + ): + litellm.completion( + model="bedrock_mantle/google.gemma-4-31b", + messages=[{"role": "user", "content": "hello"}], + aws_role_name="arn:aws:iam::000000000000:role/attributed-role", + aws_session_name="user-123", + aws_region_name="us-east-1", + ) + + assert len(credential_calls) == 1 + assert ( + credential_calls[0]["aws_role_name"] + == "arn:aws:iam::000000000000:role/attributed-role" + ) + assert credential_calls[0]["aws_session_name"] == "user-123" + assert requests[0]["headers"]["Authorization"].startswith("AWS4-HMAC-SHA256") + assert not [key for key in requests[0]["body"] if key.startswith("aws_")] + class TestBedrockMantleProjectHeader: def test_validate_environment_sets_openai_project_header(self): diff --git a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py index fddd8d09dfc..0906b39c514 100644 --- a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py +++ b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py @@ -2071,3 +2071,22 @@ async def test_anthropic_invalid_thinking_signature_retry_resigns_bedrock_reques retry_authorization = posts[1]["headers"]["Authorization"] assert retry_authorization.startswith("AWS4-HMAC-SHA256") assert retry_authorization != first_attempt_headers["Authorization"] + + +def test_aws_signing_overrides_only_fills_missing_credentials(): + from litellm.llms.custom_httpx.llm_http_handler import _aws_signing_overrides + + overrides = _aws_signing_overrides( + {"temperature": 0.2, "aws_region_name": "us-west-2"}, + { + "aws_role_name": "arn:aws:iam::000000000000:role/attributed", + "aws_session_name": "user-123", + "aws_region_name": "us-east-1", + "api_key": "not-an-aws-param", + }, + ) + + assert dict(overrides) == { + "aws_role_name": "arn:aws:iam::000000000000:role/attributed", + "aws_session_name": "user-123", + } From 2c7de60692d7a4fcd53964872d4355042d52b90b Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 12 Aug 2026 00:53:40 +0000 Subject: [PATCH 02/31] style: ruff format Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/custom_httpx/llm_http_handler.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index dfc234d08e0..cadf4c701e2 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -253,9 +253,7 @@ def _has_pre_call_deployment_hook(logging_obj: LiteLLMLoggingObj) -> bool: return False -def _aws_signing_overrides( - optional_params: Mapping[str, Any], litellm_params: Mapping[str, Any] -) -> Mapping[str, Any]: +def _aws_signing_overrides(optional_params: Mapping[str, Any], litellm_params: Mapping[str, Any]) -> Mapping[str, Any]: """AWS credential params for SigV4 signers that read them off optional_params. Only `bedrock`/`sagemaker` keep `aws_*` in optional_params: every other provider From 3275459aec0935b03b60950ba5d625854e2d6c9b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 29 Aug 2026 12:47:26 -0700 Subject: [PATCH 03/31] fix(mcp): cap tools preview and test-connection at the listing timeout and name the unreachable upstream --- .../mcp_server/rest_endpoints.py | 34 +++++--- .../mcp_server/test_rest_endpoints.py | 77 +++++++++++++++++-- 2 files changed, 96 insertions(+), 15 deletions(-) diff --git a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py index 3efb6429326..d73277f4417 100644 --- a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py +++ b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py @@ -4,10 +4,12 @@ from collections.abc import Awaitable, Callable, Mapping from datetime import datetime from typing import TYPE_CHECKING, Any, Final, Literal +import anyio import httpx from fastapi import APIRouter, Depends, HTTPException, Query, Request, status from litellm._logging import verbose_logger +from litellm.constants import MCP_TOOL_LISTING_TIMEOUT from litellm.exceptions import ( BlockedPiiEntityError, GuardrailRaisedException, @@ -68,7 +70,13 @@ _MCP_GUARDRAIL_REJECTIONS: Final = ( ) -def _connection_error_message(exc: BaseException) -> str: +def _connection_error_message(exc: BaseException, url: str | None, timeout_seconds: float) -> str: + if isinstance(exc, TimeoutError): + return ( + f"Failed to connect to MCP server: no response from {url or 'the server'} " + f"within {timeout_seconds:.0f}s. Check that the LiteLLM proxy can reach this URL " + "from its network (DNS, egress rules, firewalls) and that the server answers MCP requests." + ) if isinstance(exc, httpx.LocalProtocolError): return ( "Failed to connect to MCP server: a request header is malformed. " @@ -1136,6 +1144,7 @@ if MCP_AVAILABLE: mcp_auth_header: str | dict[str, str] | None = None, oauth2_headers: dict[str, str] | None = None, raw_headers: dict[str, str] | None = None, + timeout_seconds: float = MCP_TOOL_LISTING_TIMEOUT, ) -> Mapping[str, object]: """ Create a temporary MCP client from *request*, run *operation*, and return the result. @@ -1151,6 +1160,10 @@ if MCP_AVAILABLE: oauth2_headers: Headers extracted from the incoming request (may contain the litellm API key — must NOT be forwarded for M2M servers). raw_headers: Raw request headers forwarded for stdio env construction. + timeout_seconds: Cap on OAuth discovery, connect, handshake, and *operation* + combined. Defaults to ``MCP_TOOL_LISTING_TIMEOUT`` (30s, below common LB + timeouts) so an unreachable upstream yields this endpoint's JSON error + instead of an opaque load-balancer 504 with an empty body. Returns: The dict returned by *operation*, or an error dict on failure. @@ -1240,15 +1253,16 @@ if MCP_AVAILABLE: static_headers=request.static_headers, ) - client: Final = await global_mcp_server_manager._create_mcp_client( - server=server_model, - mcp_auth_header=mcp_auth_header, - extra_headers=merged_headers, - stdio_env=stdio_env, - cred_provider=preview_cred_provider, - ) + with anyio.fail_after(timeout_seconds): + client: Final = await global_mcp_server_manager._create_mcp_client( + server=server_model, + mcp_auth_header=mcp_auth_header, + extra_headers=merged_headers, + stdio_env=stdio_env, + cred_provider=preview_cred_provider, + ) - return await operation(client) + return await operation(client) except (KeyboardInterrupt, SystemExit, asyncio.CancelledError): raise @@ -1257,7 +1271,7 @@ if MCP_AVAILABLE: return { "status": "error", "error": True, - "message": _connection_error_message(e), + "message": _connection_error_message(e, request.url, timeout_seconds), } async def _preview_openapi_tools(spec_path: str) -> dict: diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py index ef5631218f3..51b946c11b7 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py @@ -1,4 +1,5 @@ import asyncio +import inspect import json import sys from datetime import datetime @@ -13,6 +14,7 @@ import pytest from fastapi import HTTPException from starlette.requests import Request +from litellm.constants import MCP_TOOL_LISTING_TIMEOUT from litellm.proxy._experimental.mcp_server import rest_endpoints from litellm.proxy._experimental.mcp_server.auth import ( user_api_key_auth_mcp as auth_mcp, @@ -109,6 +111,71 @@ class TestExecuteWithMcpClient: assert result["status"] == "error" assert "stack_trace" not in result + @pytest.mark.asyncio + async def test_timeout_caps_hanging_operation_and_names_url(self, monkeypatch): + async def fake_create_client(*args, **kwargs): + return object() + + monkeypatch.setattr( + rest_endpoints.global_mcp_server_manager, + "_create_mcp_client", + fake_create_client, + ) + + async def hanging_operation(client): + await asyncio.Event().wait() + + payload = NewMCPServerRequest( + server_name="example", + url="https://mcp.example.com/mcp/", + auth_type=MCPAuth.none, + ) + + result = await asyncio.wait_for( + rest_endpoints._execute_with_mcp_client(payload, hanging_operation, timeout_seconds=0.05), + timeout=5, + ) + + assert result["error"] is True + assert "https://mcp.example.com/mcp/" in result["message"] + + @pytest.mark.asyncio + async def test_timeout_covers_client_creation(self, monkeypatch): + async def hanging_create_client(*args, **kwargs): + await asyncio.Event().wait() + + monkeypatch.setattr( + rest_endpoints.global_mcp_server_manager, + "_create_mcp_client", + hanging_create_client, + ) + + async def unreached_operation(client): + return {"status": "ok"} + + payload = NewMCPServerRequest( + server_name="example", + url="https://mcp.example.com/mcp/", + auth_type=MCPAuth.none, + ) + + result = await asyncio.wait_for( + rest_endpoints._execute_with_mcp_client(payload, unreached_operation, timeout_seconds=0.05), + timeout=5, + ) + + assert result["error"] is True + assert "https://mcp.example.com/mcp/" in result["message"] + + def test_timeout_defaults_to_tool_listing_timeout(self): + default = inspect.signature(rest_endpoints._execute_with_mcp_client).parameters["timeout_seconds"].default + assert default == MCP_TOOL_LISTING_TIMEOUT + + def test_connection_error_message_timeout_names_url_and_budget(self): + message = rest_endpoints._connection_error_message(TimeoutError(), "https://api.example.com/mcp/", 30.0) + assert "https://api.example.com/mcp/" in message + assert "30s" in message + @pytest.mark.asyncio async def test_forwards_static_headers(self, monkeypatch): """Ensure static_headers are forwarded to the MCP client during test calls. @@ -2881,17 +2948,17 @@ class TestConnectionErrorMessage: secret = "Bearer sk-super-secret-token" exc = httpx.LocalProtocolError(f"Illegal header value b' {secret}'") - message = rest_endpoints._connection_error_message(exc) + message = rest_endpoints._connection_error_message(exc, "https://example.com", 30.0) assert "header" in message.lower() assert secret not in message def test_connect_error_points_at_reachability(self): - message = rest_endpoints._connection_error_message(httpx.ConnectError("All connection attempts failed")) + message = rest_endpoints._connection_error_message(httpx.ConnectError("All connection attempts failed"), "https://example.com", 30.0) assert "unreachable" in message.lower() def test_timeout_error_message(self): - message = rest_endpoints._connection_error_message(httpx.ConnectTimeout("timed out")) + message = rest_endpoints._connection_error_message(httpx.ConnectTimeout("timed out"), "https://example.com", 30.0) assert "unreachable" in message.lower() def test_http_status_error_includes_status_code(self): @@ -2901,11 +2968,11 @@ class TestConnectionErrorMessage: request=httpx.Request("POST", "http://x/"), response=response, ) - message = rest_endpoints._connection_error_message(exc) + message = rest_endpoints._connection_error_message(exc, "https://example.com", 30.0) assert "503" in message def test_unknown_error_falls_back_to_generic(self): - message = rest_endpoints._connection_error_message(RuntimeError("weird")) + message = rest_endpoints._connection_error_message(RuntimeError("weird"), "https://example.com", 30.0) assert "weird" not in message assert "proxy logs" in message.lower() From f4b5449c6a65cf167658fd5bb32695abda9633a2 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 29 Aug 2026 12:48:04 -0700 Subject: [PATCH 04/31] fix(openai_like): strip cache_control ttl before forwarding /v1/messages to non-Anthropic providers --- litellm/llms/anthropic/common_utils.py | 33 +++++ litellm/llms/openai_like/README.md | 5 +- .../openai_like/messages/transformation.py | 38 +++++- ..._like_anthropic_messages_transformation.py | 121 ++++++++++++++++++ 4 files changed, 195 insertions(+), 2 deletions(-) diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index 681a8397f66..695e3a313ef 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -1302,6 +1302,39 @@ def flatten_unencrypted_web_search_results_in_anthropic_messages( # mutable-ok: return [_flatten_web_search_results_in_message(m) for m in messages] # mutable-ok: JSON wire format +def _normalized_cache_control(cache_control: dict) -> dict: # mutable-ok: as sibling sanitizers + cache_type: Final = cache_control.get("type") + return {"type": cache_type if isinstance(cache_type, str) else "ephemeral"} # mutable-ok: JSON wire format + + +def _normalize_cache_control_value(value: object) -> object: + if isinstance(value, dict): + return normalize_cache_control_in_anthropic_payload(value) + if isinstance(value, list): + return [_normalize_cache_control_value(item) for item in value] # mutable-ok: JSON wire format + return value + + +def normalize_cache_control_in_anthropic_payload(payload: dict) -> dict: # mutable-ok: as sibling sanitizers + """ + Return a copy of an Anthropic /v1/messages payload with every + ``cache_control`` entry reduced to ``{"type": }``, + recursing through message content blocks, system blocks, and tools. + + Anthropic itself accepts prompt-caching extensions such as ``ttl``, but + strict non-Anthropic implementations of the Messages API validate the field + literally and reject the whole request (``cache_control.ttl: 1h is not + supported``, ``cache_control.type is required``), which 400s clients like + Claude Code that always send cache hints. Non-dict ``cache_control`` values + are dropped entirely. The caller's payload is never mutated. + """ + return { # mutable-ok: JSON wire format, as sibling sanitizers + key: _normalized_cache_control(value) if key == "cache_control" else _normalize_cache_control_value(value) + for key, value in payload.items() + if key != "cache_control" or isinstance(value, dict) + } + + def process_anthropic_headers(headers: httpx.Headers | dict) -> dict: openai_headers: Final = {} if "anthropic-ratelimit-requests-limit" in headers: diff --git a/litellm/llms/openai_like/README.md b/litellm/llms/openai_like/README.md index e9aaafe48a1..e1409b81c35 100644 --- a/litellm/llms/openai_like/README.md +++ b/litellm/llms/openai_like/README.md @@ -54,7 +54,10 @@ That's it! The provider will be automatically loaded and available. "constraints": { "temperature_max": 1.0, "temperature_min": 0.0, - "temperature_min_with_n_gt_1": 0.3 + "temperature_min_with_n_gt_1": 0.3, + // /v1/messages providers only: keep Anthropic cache_control extensions + // such as ttl instead of stripping them down to {"type": ...} + "cache_control_ttl": true }, // Optional: Special handling flags diff --git a/litellm/llms/openai_like/messages/transformation.py b/litellm/llms/openai_like/messages/transformation.py index 11dc236064d..29973fe2101 100644 --- a/litellm/llms/openai_like/messages/transformation.py +++ b/litellm/llms/openai_like/messages/transformation.py @@ -1,11 +1,13 @@ from typing import Any, Final import litellm +from litellm.llms.anthropic.common_utils import normalize_cache_control_in_anthropic_payload from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( AnthropicMessagesConfig, ) from litellm.llms.openai_like.json_loader import SimpleProviderConfig from litellm.secret_managers.main import get_secret_str +from litellm.types.router import GenericLiteLLMParams DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01" @@ -19,7 +21,9 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): ``"/v1/messages"``. The inbound Anthropic payload (system, cache_control, thinking, tools, ...) is forwarded essentially unchanged to ``{api_base}/v1/messages``, so Anthropic-only features that the - Anthropic->OpenAI translation would otherwise drop are preserved. Response + Anthropic->OpenAI translation would otherwise drop are preserved. The one + exception is ``cache_control``, whose Anthropic-only extensions (``ttl``) + are stripped unless ``supports_cache_control_ttl`` says otherwise. Response parsing and streaming are inherited from the native Anthropic config. """ @@ -53,6 +57,35 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): def should_filter_anthropic_beta_headers(self) -> bool: return False + def supports_cache_control_ttl(self) -> bool: + return False + + def transform_anthropic_messages_request( + self, + model: str, + messages: list[dict], # mutable-ok: matches dict-typed base signature + anthropic_messages_optional_request_params: dict, # mutable-ok: matches dict-typed base signature + litellm_params: GenericLiteLLMParams, + headers: dict, # mutable-ok: matches dict-typed base signature + ) -> dict: # mutable-ok: matches dict-typed base signature + """ + Anthropic ignores prompt-caching hints it cannot honor, but strict + non-Anthropic implementations of the Messages API 400 the whole request + on Anthropic-only ``cache_control`` extensions (``cache_control.ttl: 1h + is not supported``), so unless the provider declares ttl support the + hints are reduced to their portable ``{"type": ...}`` core. + """ + request: Final = super().transform_anthropic_messages_request( + model=model, + messages=messages, + anthropic_messages_optional_request_params=anthropic_messages_optional_request_params, + litellm_params=litellm_params, + headers=headers, + ) + if self.supports_cache_control_ttl(): + return request + return normalize_cache_control_in_anthropic_payload(request) + def get_complete_url( self, api_base: str | None, @@ -91,6 +124,9 @@ class JSONProviderAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig): def should_strip_billing_metadata(self) -> bool: return True + def supports_cache_control_ttl(self) -> bool: + return bool(self._provider.constraints.get("cache_control_ttl")) + def _resolve_api_key(self, api_key: str | None) -> str | None: return api_key or get_secret_str(self._provider.api_key_env) or litellm.api_key diff --git a/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py index 33e677b000e..2cdd969b00f 100644 --- a/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py +++ b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py @@ -317,3 +317,124 @@ def test_json_provider_messages_config_probes_capabilities_under_provider_slug() ) assert JSONProviderAnthropicMessagesConfig(provider).custom_llm_provider == "exampleprovider" assert OpenAILikeAnthropicMessagesConfig().custom_llm_provider == "anthropic" + + +def _cache_control_request_params() -> tuple[list, dict]: + messages = [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "write a regex for a US phone number", + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + } + ], + } + ] + optional_params = { + "max_tokens": 256, + "system": [ + { + "type": "text", + "text": "You are Claude Code.", + "cache_control": {"type": "ephemeral", "ttl": "5m"}, + } + ], + "tools": [ + { + "name": "lookup", + "input_schema": {"type": "object"}, + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + } + ], + } + return messages, optional_params + + +def test_request_strips_cache_control_ttl_everywhere(config): + """Regression: Claude Code always sends ``cache_control: {type: ephemeral, + ttl: 1h}``, and strict non-Anthropic /v1/messages validators 400 the whole + request on the ttl extension (``cache_control.ttl: 1h is not supported``).""" + messages, optional_params = _cache_control_request_params() + + payload = config.transform_anthropic_messages_request( + model="some-model", + messages=messages, + anthropic_messages_optional_request_params=optional_params, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert payload["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral"} + assert payload["system"][0]["cache_control"] == {"type": "ephemeral"} + assert payload["tools"][0]["cache_control"] == {"type": "ephemeral"} + assert messages[0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"} + + +def test_request_defaults_missing_cache_control_type_and_drops_non_dict(config): + payload = config.transform_anthropic_messages_request( + model="some-model", + messages=[ + { + "role": "user", + "content": [ + {"type": "text", "text": "a", "cache_control": {"ttl": "1h"}}, + {"type": "text", "text": "b", "cache_control": None}, + ], + } + ], + anthropic_messages_optional_request_params={"max_tokens": 64}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + blocks = payload["messages"][0]["content"] + assert blocks[0]["cache_control"] == {"type": "ephemeral"} + assert "cache_control" not in blocks[1] + + +def test_native_anthropic_config_keeps_cache_control_ttl(): + """Anthropic itself accepts ttl, so the normalization must stay scoped to + the OpenAI-like passthrough and never reach the native Anthropic path.""" + from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( + AnthropicMessagesConfig, + ) + + messages, optional_params = _cache_control_request_params() + payload = AnthropicMessagesConfig().transform_anthropic_messages_request( + model="claude-sonnet-4-20250514", + messages=messages, + anthropic_messages_optional_request_params=optional_params, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert payload["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"} + assert payload["system"][0]["cache_control"] == {"type": "ephemeral", "ttl": "5m"} + + +def test_json_provider_constraint_opts_into_cache_control_ttl(): + from litellm.llms.openai_like.json_loader import SimpleProviderConfig + from litellm.llms.openai_like.messages.transformation import ( + JSONProviderAnthropicMessagesConfig, + ) + + base_data = {"base_url": "https://api.example.com/v1", "api_key_env": "EXAMPLE_API_KEY"} + strict = JSONProviderAnthropicMessagesConfig(SimpleProviderConfig(slug="strictprov", data=base_data)) + lenient = JSONProviderAnthropicMessagesConfig( + SimpleProviderConfig(slug="lenientprov", data={**base_data, "constraints": {"cache_control_ttl": True}}) + ) + + def transform(provider_config): + messages, optional_params = _cache_control_request_params() + return provider_config.transform_anthropic_messages_request( + model="some-model", + messages=messages, + anthropic_messages_optional_request_params=optional_params, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert transform(strict)["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral"} + assert transform(lenient)["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"} From d3268e4e184f8b7cde6cce4473e14b1acedfc751 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 29 Aug 2026 13:26:04 -0700 Subject: [PATCH 05/31] test(mcp): wrap over-long connection error message calls --- .../proxy/_experimental/mcp_server/test_rest_endpoints.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py index 51b946c11b7..e66101f6177 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_rest_endpoints.py @@ -2954,11 +2954,15 @@ class TestConnectionErrorMessage: assert secret not in message def test_connect_error_points_at_reachability(self): - message = rest_endpoints._connection_error_message(httpx.ConnectError("All connection attempts failed"), "https://example.com", 30.0) + message = rest_endpoints._connection_error_message( + httpx.ConnectError("All connection attempts failed"), "https://example.com", 30.0 + ) assert "unreachable" in message.lower() def test_timeout_error_message(self): - message = rest_endpoints._connection_error_message(httpx.ConnectTimeout("timed out"), "https://example.com", 30.0) + message = rest_endpoints._connection_error_message( + httpx.ConnectTimeout("timed out"), "https://example.com", 30.0 + ) assert "unreachable" in message.lower() def test_http_status_error_includes_status_code(self): From 0baf376efd691e139902ca7933b83666ea459a41 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 29 Aug 2026 14:07:05 -0700 Subject: [PATCH 06/31] fix(openai_like): scope cache_control normalization to Messages API locations Rewrite the sanitizer without recursion (the code-quality gate rejects new recursive functions) and only touch cache_control where the Messages API defines it: the request, system blocks, tools, message content blocks, and tool_result content. Application data such as tool_use.input and tool input_schema is left untouched even when it contains a cache_control key --- litellm/llms/anthropic/common_utils.py | 78 +++++++++++++++---- ..._like_anthropic_messages_transformation.py | 62 +++++++++++++++ 2 files changed, 124 insertions(+), 16 deletions(-) diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index 695e3a313ef..1ef14362601 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -1302,37 +1302,83 @@ def flatten_unencrypted_web_search_results_in_anthropic_messages( # mutable-ok: return [_flatten_web_search_results_in_message(m) for m in messages] # mutable-ok: JSON wire format -def _normalized_cache_control(cache_control: dict) -> dict: # mutable-ok: as sibling sanitizers +def _normalized_cache_control(cache_control: object) -> dict[str, str] | None: # mutable-ok: JSON wire format + if not isinstance(cache_control, Mapping): + return None cache_type: Final = cache_control.get("type") return {"type": cache_type if isinstance(cache_type, str) else "ephemeral"} # mutable-ok: JSON wire format -def _normalize_cache_control_value(value: object) -> object: - if isinstance(value, dict): - return normalize_cache_control_in_anthropic_payload(value) - if isinstance(value, list): - return [_normalize_cache_control_value(item) for item in value] # mutable-ok: JSON wire format - return value +def _with_portable_cache_control(block: Mapping[str, object]) -> dict[str, object]: # mutable-ok: JSON wire format + if "cache_control" not in block: + return dict(block) # mutable-ok: JSON wire format + normalized: Final = _normalized_cache_control(block["cache_control"]) + rest: Final = {key: value for key, value in block.items() if key != "cache_control"} # mutable-ok: JSON wire format + return rest if normalized is None else {**rest, "cache_control": normalized} # mutable-ok: JSON wire format -def normalize_cache_control_in_anthropic_payload(payload: dict) -> dict: # mutable-ok: as sibling sanitizers +def _with_portable_cache_control_in_blocks(blocks: object) -> object: + if isinstance(blocks, str) or not isinstance(blocks, Sequence): + return blocks + return [ # mutable-ok: JSON wire format + _with_portable_cache_control(block) if isinstance(block, Mapping) else block for block in blocks + ] + + +def _with_portable_cache_control_in_content_block(block: object) -> object: + if not isinstance(block, Mapping): + return block + portable: Final = _with_portable_cache_control(block) + if portable.get("type") != "tool_result" or "content" not in portable: + return portable + return { # mutable-ok: JSON wire format + **portable, + "content": _with_portable_cache_control_in_blocks(portable["content"]), + } + + +def _with_portable_cache_control_in_message(message: object) -> object: + if not isinstance(message, Mapping) or "content" not in message: + return message + content: Final = message["content"] + if isinstance(content, str) or not isinstance(content, Sequence): + return message + return { # mutable-ok: JSON wire format + **message, + "content": [_with_portable_cache_control_in_content_block(block) for block in content], + } + + +def normalize_cache_control_in_anthropic_payload( # mutable-ok: JSON wire format + payload: Mapping[str, object], +) -> dict[str, object]: """ Return a copy of an Anthropic /v1/messages payload with every - ``cache_control`` entry reduced to ``{"type": }``, - recursing through message content blocks, system blocks, and tools. + ``cache_control`` entry reduced to ``{"type": }`` + at the places the Messages API defines it: the request itself, system + blocks, tools, message content blocks, and ``tool_result`` content blocks. + Application data such as ``tool_use.input`` and tool ``input_schema`` is + never touched, even when it happens to contain a ``cache_control`` key. Anthropic itself accepts prompt-caching extensions such as ``ttl``, but strict non-Anthropic implementations of the Messages API validate the field literally and reject the whole request (``cache_control.ttl: 1h is not supported``, ``cache_control.type is required``), which 400s clients like - Claude Code that always send cache hints. Non-dict ``cache_control`` values - are dropped entirely. The caller's payload is never mutated. + Claude Code that send cache hints. Non-dict ``cache_control`` values are + dropped entirely. The caller's payload is never mutated. """ - return { # mutable-ok: JSON wire format, as sibling sanitizers - key: _normalized_cache_control(value) if key == "cache_control" else _normalize_cache_control_value(value) - for key, value in payload.items() - if key != "cache_control" or isinstance(value, dict) + portable: Final = _with_portable_cache_control(payload) + scoped: Final = { # mutable-ok: JSON wire format + key: ( + _with_portable_cache_control_in_blocks(value) + if key in ("system", "tools") + else [_with_portable_cache_control_in_message(message) for message in value] + if key == "messages" and isinstance(value, Sequence) and not isinstance(value, str) + else value + ) + for key, value in portable.items() } + return scoped def process_anthropic_headers(headers: httpx.Headers | dict) -> dict: diff --git a/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py index 2cdd969b00f..d325492914e 100644 --- a/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py +++ b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py @@ -438,3 +438,65 @@ def test_json_provider_constraint_opts_into_cache_control_ttl(): assert transform(strict)["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral"} assert transform(lenient)["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"} + + +def test_request_strips_ttl_only_where_the_messages_api_defines_cache_control(config): + """Regression: the sanitizer must only touch ``cache_control`` where the + Messages API defines it (request, system, tools, content blocks, tool_result + content), never application data such as ``tool_use.input`` or a tool's + ``input_schema`` that happens to contain a ``cache_control`` key.""" + tool_input = {"cache_control": {"type": "ephemeral", "ttl": "1h"}, "query": "x"} + input_schema = { + "type": "object", + "properties": {"cache_control": {"type": "string", "ttl": "1h"}}, + } + messages = [ + { + "role": "assistant", + "content": [{"type": "tool_use", "id": "toolu_1", "name": "lookup", "input": tool_input}], + }, + { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "toolu_1", + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + "content": [ + {"type": "text", "text": "result", "cache_control": {"type": "ephemeral", "ttl": "1h"}} + ], + }, + {"type": "text", "text": "plain string content stays", "cache_control": {"ttl": "1h"}}, + ], + }, + {"role": "user", "content": "a plain string message"}, + ] + optional_params = { + "max_tokens": 64, + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + "tools": [ + { + "name": "lookup", + "input_schema": input_schema, + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + } + ], + } + + payload = config.transform_anthropic_messages_request( + model="some-model", + messages=messages, + anthropic_messages_optional_request_params=optional_params, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert payload["cache_control"] == {"type": "ephemeral"} + assert payload["tools"][0]["cache_control"] == {"type": "ephemeral"} + assert payload["tools"][0]["input_schema"] == input_schema + assert payload["messages"][0]["content"][0]["input"] == tool_input + tool_result = payload["messages"][1]["content"][0] + assert tool_result["cache_control"] == {"type": "ephemeral"} + assert tool_result["content"][0]["cache_control"] == {"type": "ephemeral"} + assert payload["messages"][1]["content"][1]["cache_control"] == {"type": "ephemeral"} + assert payload["messages"][2] == {"role": "user", "content": "a plain string message"} From c32eb41aad3b7b087c7d0023a71876d3dea6511d Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 29 Aug 2026 14:17:13 -0700 Subject: [PATCH 07/31] feat(openai_like): let a passthrough deployment keep cache_control ttl via model_info.cache_control_ttl The supported_endpoints passthrough had no way to keep ttl for an upstream that honors it, so the deployment now opts in with model_info.cache_control_ttl: true, injected into the config the same way the providers.json constraint is for JSON providers --- .../messages/handler.py | 8 +- .../openai_like/messages/transformation.py | 16 +-- ...erimental_pass_through_messages_handler.py | 97 ++++++++++--------- ..._like_anthropic_messages_transformation.py | 17 ++++ 4 files changed, 84 insertions(+), 54 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 69985bcdaa3..b82903d6f87 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -99,6 +99,10 @@ def _deployment_passes_through_anthropic_messages(model_info: object) -> bool: return isinstance(supported_endpoints, (list, tuple)) and "/v1/messages" in supported_endpoints +def _deployment_supports_cache_control_ttl(model_info: object) -> bool: + return isinstance(model_info, dict) and model_info.get("cache_control_ttl") is True + + ####### ENVIRONMENT VARIABLES ################### # Initialize any necessary instances or variables here base_llm_http_handler = BaseLLMHTTPHandler() @@ -568,7 +572,9 @@ def anthropic_messages_handler( OpenAILikeAnthropicMessagesConfig, ) - anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig() + anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig( + cache_control_ttl=_deployment_supports_cache_control_ttl(kwargs.get("model_info")), + ) if anthropic_messages_provider_config is None: # Route to Responses API for OpenAI / Azure, chat/completions for everything else. if _should_route_to_responses_api(custom_llm_provider, original_model, model): diff --git a/litellm/llms/openai_like/messages/transformation.py b/litellm/llms/openai_like/messages/transformation.py index 29973fe2101..ac99617521c 100644 --- a/litellm/llms/openai_like/messages/transformation.py +++ b/litellm/llms/openai_like/messages/transformation.py @@ -23,10 +23,15 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): ``{api_base}/v1/messages``, so Anthropic-only features that the Anthropic->OpenAI translation would otherwise drop are preserved. The one exception is ``cache_control``, whose Anthropic-only extensions (``ttl``) - are stripped unless ``supports_cache_control_ttl`` says otherwise. Response - parsing and streaming are inherited from the native Anthropic config. + are stripped unless the deployment opts in with + ``model_info.cache_control_ttl: true``. Response parsing and streaming are + inherited from the native Anthropic config. """ + def __init__(self, cache_control_ttl: bool = False) -> None: + super().__init__() + self._cache_control_ttl: Final = cache_control_ttl + def validate_anthropic_messages_environment( self, headers: dict[str, str], @@ -58,7 +63,7 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): return False def supports_cache_control_ttl(self) -> bool: - return False + return self._cache_control_ttl def transform_anthropic_messages_request( self, @@ -114,7 +119,7 @@ class JSONProviderAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig): """ def __init__(self, provider: SimpleProviderConfig): - super().__init__() + super().__init__(cache_control_ttl=bool(provider.constraints.get("cache_control_ttl"))) self._provider = provider @property @@ -124,9 +129,6 @@ class JSONProviderAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig): def should_strip_billing_metadata(self) -> bool: return True - def supports_cache_control_ttl(self) -> bool: - return bool(self._provider.constraints.get("cache_control_ttl")) - def _resolve_api_key(self, api_key: str | None) -> str | None: return api_key or get_secret_str(self._provider.api_key_env) or litellm.api_key diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index ad4c3d6bfbb..e819433c269 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -296,21 +296,15 @@ async def test_bedrock_converse_budget_tokens_preserved(): mock_acompletion.assert_called_once() call_kwargs = mock_acompletion.call_args.kwargs - print( - "acompletion call kwargs: ", json.dumps(call_kwargs, indent=4, default=str) - ) + print("acompletion call kwargs: ", json.dumps(call_kwargs, indent=4, default=str)) # Verify thinking parameter is passed through with budget_tokens preserved thinking_param = call_kwargs.get("thinking") - assert ( - thinking_param is not None - ), "thinking parameter should be passed to acompletion" - assert ( - thinking_param.get("type") == "enabled" - ), "thinking.type should be 'enabled'" - assert ( - thinking_param.get("budget_tokens") == 1024 - ), f"thinking.budget_tokens should be 1024, but got {thinking_param.get('budget_tokens')}" + assert thinking_param is not None, "thinking parameter should be passed to acompletion" + assert thinking_param.get("type") == "enabled", "thinking.type should be 'enabled'" + assert thinking_param.get("budget_tokens") == 1024, ( + f"thinking.budget_tokens should be 1024, but got {thinking_param.get('budget_tokens')}" + ) def test_openai_model_with_thinking_converts_to_reasoning(): @@ -342,23 +336,18 @@ def test_openai_model_with_thinking_converts_to_reasoning(): call_kwargs = mock_responses.call_args.kwargs # Verify reasoning is set (converted from thinking) - assert ( - "reasoning" in call_kwargs - ), "reasoning should be passed to litellm.responses" + assert "reasoning" in call_kwargs, "reasoning should be passed to litellm.responses" # budget_tokens=1024 -> effort="low" (at the LOW budget threshold) # reasoning_auto_summary is False by default, so no summary key expected_reasoning = {"effort": "low"} assert call_kwargs["reasoning"] == expected_reasoning, ( - f"reasoning should be {expected_reasoning} for budget_tokens=1024, " - f"got {call_kwargs.get('reasoning')}" + f"reasoning should be {expected_reasoning} for budget_tokens=1024, got {call_kwargs.get('reasoning')}" ) assert "summary" not in call_kwargs["reasoning"] # Verify thinking is NOT passed directly to the Responses API - assert ( - "thinking" not in call_kwargs - ), "thinking should NOT be passed directly to litellm.responses" + assert "thinking" not in call_kwargs, "thinking should NOT be passed directly to litellm.responses" class TestThinkingParameterTransformation: @@ -411,9 +400,7 @@ class TestThinkingParameterTransformation: thinking=thinking, model="openai/gpt-5.2", ) - assert result == { - "reasoning_effort": {"effort": "high", "summary": "detailed"} - } + assert result == {"reasoning_effort": {"effort": "high", "summary": "detailed"}} finally: litellm.reasoning_auto_summary = original @@ -611,9 +598,9 @@ class TestThinkingSummaryPreservation: mock_responses.assert_called_once() call_kwargs = mock_responses.call_args.kwargs reasoning = call_kwargs["reasoning"] - assert ( - reasoning["summary"] == "concise" - ), f"Expected summary='concise', got summary='{reasoning.get('summary')}'" + assert reasoning["summary"] == "concise", ( + f"Expected summary='concise', got summary='{reasoning.get('summary')}'" + ) def test_responses_adapter_preserves_summary(self): """translate_thinking_to_reasoning should include summary when user provides it.""" @@ -622,9 +609,7 @@ class TestThinkingSummaryPreservation: ) thinking = {"type": "enabled", "budget_tokens": 5000, "summary": "concise"} - result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning( - thinking - ) + result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(thinking) assert result == {"effort": "high", "summary": "concise"} def test_responses_adapter_no_summary_by_default(self): @@ -638,11 +623,7 @@ class TestThinkingSummaryPreservation: try: litellm.reasoning_auto_summary = False thinking = {"type": "enabled", "budget_tokens": 5000} - result = ( - LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning( - thinking - ) - ) + result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(thinking) assert result == {"effort": "high"} assert result is not None and "summary" not in result finally: @@ -659,9 +640,7 @@ class TestThinkingSummaryPreservation: thinking=thinking, model="openai/gpt-5.2", ) - assert result == { - "reasoning_effort": {"effort": "high", "summary": "concise"} - } + assert result == {"reasoning_effort": {"effort": "high", "summary": "concise"}} def test_translate_thinking_for_model_disabled_stays_plain_string_when_auto_summary_enabled(self): """Disabled thinking must stay a plain string even when reasoning_auto_summary is on.""" @@ -807,9 +786,7 @@ def test_presanitized_flag_not_leaked_to_provider_params(): def fake_base_handler(*args, **kwargs): captured.update(kwargs) - captured["optional"] = kwargs.get( - "anthropic_messages_optional_request_params", {} - ) + captured["optional"] = kwargs.get("anthropic_messages_optional_request_params", {}) return "stub" with patch.object( @@ -974,6 +951,38 @@ def test_gate_passthrough_skipped_when_only_chat_completions_supported(monkeypat assert "config" not in captured +@pytest.mark.parametrize( + "model_info, expected_ttl_support", + [ + ({"supported_endpoints": ["/v1/messages"]}, False), + ({"supported_endpoints": ["/v1/messages"], "cache_control_ttl": True}, True), + ({"supported_endpoints": ["/v1/messages"], "cache_control_ttl": "yes"}, False), + ], +) +def test_gate_passthrough_forwards_cache_control_ttl_only_when_deployment_opts_in( + monkeypatch, model_info, expected_ttl_support +): + """The passthrough config strips cache_control.ttl unless the deployment sets + model_info.cache_control_ttl to exactly true.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + captured, _ = _gate_stubs(monkeypatch) + + result = anthropic_messages_handler( + max_tokens=100, + messages=[{"role": "user", "content": "Hello"}], + model="openai/some-model", + api_key="sk-test", + api_base="https://host/v1", + model_info=model_info, + ) + + assert result == "native-passthrough" + assert captured["config"].supports_cache_control_ttl() is expected_ttl_support + + def test_first_party_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_flag(): """Regional and provider-prefixed Claude 4.8+/5 entries carry ``supports_mid_conversation_system``, but the bare first-party keys @@ -987,9 +996,7 @@ def test_first_party_claude_4_8_plus_cost_map_entries_carry_mid_conversation_sys import litellm - cost_map_path = os.path.join( - os.path.dirname(litellm.__file__), "model_prices_and_context_window_backup.json" - ) + cost_map_path = os.path.join(os.path.dirname(litellm.__file__), "model_prices_and_context_window_backup.json") with open(cost_map_path) as f: cost_map = json.load(f) rules = cost_map["fallback_generalizations"]["rules"] @@ -1028,9 +1035,7 @@ def test_first_party_claude_4_8_plus_cost_map_entries_carry_mid_conversation_sys ("perplexity/sonar", "sonar", "https://api.perplexity.ai/chat/completions"), ], ) -async def test_messages_strips_provider_prefix_exactly_once( - requested_model, expected_wire_model, expected_url -): +async def test_messages_strips_provider_prefix_exactly_once(requested_model, expected_wire_model, expected_url): """ BerriAI/litellm#37716: only the leading provider segment may be stripped on the way upstream. diff --git a/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py index d325492914e..e33b03afdff 100644 --- a/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py +++ b/tests/test_litellm/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py @@ -414,6 +414,23 @@ def test_native_anthropic_config_keeps_cache_control_ttl(): assert payload["system"][0]["cache_control"] == {"type": "ephemeral", "ttl": "5m"} +def test_deployment_opt_in_keeps_cache_control_ttl(): + config = OpenAILikeAnthropicMessagesConfig(cache_control_ttl=True) + payload = config.transform_anthropic_messages_request( + model="some-model", + messages=[ + { + "role": "user", + "content": [{"type": "text", "text": "hi", "cache_control": {"type": "ephemeral", "ttl": "1h"}}], + } + ], + anthropic_messages_optional_request_params={"max_tokens": 16}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert payload["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"} + + def test_json_provider_constraint_opts_into_cache_control_ttl(): from litellm.llms.openai_like.json_loader import SimpleProviderConfig from litellm.llms.openai_like.messages.transformation import ( From d4fc54a11d1ac18f10c33741b760b99716faac93 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 08:25:42 +0000 Subject: [PATCH 08/31] chore(techdebt): clear fresh debt from the 2026-08-31 window Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- basedpyright-code-budget.json | 6 +-- litellm/llms/gigachat/authenticator.py | 28 +++++--------- litellm/llms/gigachat/chat/streaming.py | 1 - litellm/llms/gigachat/chat/transformation.py | 38 +++++-------------- .../llms/gigachat/embedding/transformation.py | 14 ++----- .../gigachat/passthrough/transformation.py | 2 - litellm/llms/gigachat/utils.py | 1 - litellm/passthrough/main.py | 3 -- .../llm_passthrough_endpoints.py | 4 +- .../router_strategy/test_complexity_router.py | 6 ++- type-discipline-budget.json | 8 ++-- 11 files changed, 35 insertions(+), 76 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index a07b9352659..d84aacfeaf0 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5607 }, "reportMissingTypeArgument": { - "limit": 15310 + "limit": 15308 }, "reportMissingTypeStubs": { "limit": 40 @@ -105,13 +105,13 @@ "limit": 109 }, "reportUnknownMemberType": { - "limit": 38368 + "limit": 38367 }, "reportUnknownParameterType": { "limit": 19633 }, "reportUnknownVariableType": { - "limit": 29908 + "limit": 29906 }, "reportUnnecessaryCast": { "limit": 111 diff --git a/litellm/llms/gigachat/authenticator.py b/litellm/llms/gigachat/authenticator.py index d6b217d5746..73086ba395b 100644 --- a/litellm/llms/gigachat/authenticator.py +++ b/litellm/llms/gigachat/authenticator.py @@ -8,6 +8,7 @@ Based on official GigaChat SDK authentication flow. import time import uuid from collections.abc import Mapping +from types import MappingProxyType from typing import Final import httpx @@ -32,8 +33,8 @@ GIGACHAT_SCOPE: Final = "GIGACHAT_API_PERS" # Token expiry buffer in milliseconds (refresh token 60s before expiry) TOKEN_EXPIRY_BUFFER_MS: Final = 60000 -# Cache for access tokens _token_cache: Final = InMemoryCache() +_NO_LITELLM_PARAMS: Final[Mapping[str, object]] = MappingProxyType({}) class GigaChatAuthError(BaseLLMException): @@ -80,10 +81,9 @@ def get_access_token( Raises: GigaChatAuthError: If authentication fails """ - if not litellm_params: - litellm_params = {} # mutable-ok: empty dict default; rebind-ok: provide default + params: Final = litellm_params or _NO_LITELLM_PARAMS - access_token: Final = litellm_params.get("gigachat_access_token") or get_secret_str("GIGACHAT_ACCESS_TOKEN") + access_token: Final = params.get("gigachat_access_token") or get_secret_str("GIGACHAT_ACCESS_TOKEN") if access_token: return access_token @@ -94,24 +94,20 @@ def get_access_token( message="GigaChat credentials not provided. Set GIGACHAT_CREDENTIALS or GIGACHAT_API_KEY environment variable.", ) - effective_scope: Final = scope or litellm_params.get("gigachat_scope") or _get_scope() - effective_auth_url: Final = auth_url or litellm_params.get("gigachat_auth_url") or _get_auth_url() + effective_scope: Final = scope or params.get("gigachat_scope") or _get_scope() + effective_auth_url: Final = auth_url or params.get("gigachat_auth_url") or _get_auth_url() - # Check cache cache_key: Final = f"gigachat_token:{effective_credentials[:16]}" cached: Final = _token_cache.get_cache(cache_key) if cached: _token, _expires_at = cached - # Check if token is still valid (with buffer) if time.time() * 1000 < _expires_at - TOKEN_EXPIRY_BUFFER_MS: verbose_logger.debug("Using cached GigaChat access token") return _token - # Request new token new_token, new_expires_at = _request_token_sync(effective_credentials, effective_scope, effective_auth_url) # pyright: ignore[reportArgumentType] # credential keys may be broader than str if new_expires_at: - # Cache token ttl_seconds: Final = max(0, (new_expires_at - TOKEN_EXPIRY_BUFFER_MS - time.time() * 1000) / 1000) if ttl_seconds > 0: _token_cache.set_cache(cache_key, (new_token, new_expires_at), ttl=ttl_seconds) @@ -126,10 +122,9 @@ async def get_access_token_async( litellm_params: Mapping[str, object] | None = None, ) -> str: """Async version of get_access_token.""" - if not litellm_params: - litellm_params = {} # mutable-ok: empty dict default; rebind-ok: provide default + params: Final = litellm_params or _NO_LITELLM_PARAMS - access_token: Final = litellm_params.get("gigachat_access_token") or get_secret_str("GIGACHAT_ACCESS_TOKEN") + access_token: Final = params.get("gigachat_access_token") or get_secret_str("GIGACHAT_ACCESS_TOKEN") if access_token: return access_token @@ -140,10 +135,9 @@ async def get_access_token_async( message="GigaChat credentials not provided. Set GIGACHAT_CREDENTIALS or GIGACHAT_API_KEY environment variable.", ) - effective_scope: Final = scope or litellm_params.get("gigachat_scope") or _get_scope() - effective_auth_url: Final = auth_url or litellm_params.get("gigachat_auth_url") or _get_auth_url() + effective_scope: Final = scope or params.get("gigachat_scope") or _get_scope() + effective_auth_url: Final = auth_url or params.get("gigachat_auth_url") or _get_auth_url() - # Check cache cache_key: Final = f"gigachat_token:{effective_credentials[:16]}" cached: Final = _token_cache.get_cache(cache_key) if cached: @@ -152,11 +146,9 @@ async def get_access_token_async( verbose_logger.debug("Using cached GigaChat access token") return _token - # Request new token new_token, new_expires_at = await _request_token_async(effective_credentials, effective_scope, effective_auth_url) # pyright: ignore[reportArgumentType] # credential keys may be broader than str if new_expires_at: - # Cache token ttl_seconds: Final = max(0, (new_expires_at - TOKEN_EXPIRY_BUFFER_MS - time.time() * 1000) / 1000) if ttl_seconds > 0: _token_cache.set_cache(cache_key, (new_token, new_expires_at), ttl=ttl_seconds) diff --git a/litellm/llms/gigachat/chat/streaming.py b/litellm/llms/gigachat/chat/streaming.py index 2875b30232e..0a4cbd8e520 100644 --- a/litellm/llms/gigachat/chat/streaming.py +++ b/litellm/llms/gigachat/chat/streaming.py @@ -52,7 +52,6 @@ class GigaChatModelResponseIterator: tool_use: ChatCompletionToolCallChunk | None = None # rebind-ok: conditionally assigned on function_call finish_reason: str | None = chunk_finish_reason - # Handle function_call in stream raw_function_call: Final = delta.get("function_call") if chunk_finish_reason == "function_call" and isinstance(raw_function_call, Mapping) and raw_function_call: func_call: Final[Mapping[str, object]] = raw_function_call diff --git a/litellm/llms/gigachat/chat/transformation.py b/litellm/llms/gigachat/chat/transformation.py index 8f23c5175ec..991a93ccb21 100644 --- a/litellm/llms/gigachat/chat/transformation.py +++ b/litellm/llms/gigachat/chat/transformation.py @@ -111,11 +111,9 @@ class GigaChatConfig(BaseConfig): """ Set up headers with OAuth token. """ - # Get access token credentials: Final = api_key or get_secret_str("GIGACHAT_CREDENTIALS") or get_secret_str("GIGACHAT_API_KEY") access_token: Final = get_access_token(credentials=credentials, litellm_params=litellm_params) - # Store credentials for image uploads self._current_credentials = credentials self._current_api_base = api_base @@ -208,18 +206,16 @@ class GigaChatConfig(BaseConfig): def _convert_tools_to_functions(self, tools: Sequence) -> Sequence[dict]: """Convert OpenAI tools format to GigaChat functions format.""" - functions: Final[list[dict]] = [] # mutable-ok: accumulator for building functions list - for tool in tools: - if isinstance(tool, dict) and tool.get("type") == "function": - func = tool.get("function", {}) - functions.append( - { - "name": func.get("name", ""), - "description": func.get("description", ""), - "parameters": func.get("parameters", {}), - } - ) - return functions + return [ + { + "name": function.get("name", ""), + "description": function.get("description", ""), + "parameters": function.get("parameters", {}), + } + for function in ( + tool.get("function", {}) for tool in tools if isinstance(tool, dict) and tool.get("type") == "function" + ) + ] def _map_tool_choice(self, tool_choice: str | Mapping[str, object]) -> str | Mapping[str, object] | None: """ @@ -299,7 +295,6 @@ class GigaChatConfig(BaseConfig): if part.get("type") == "text": texts.append(part.get("text", "")) elif part.get("type") == "image_url": - # Extract image URL and upload to GigaChat image_url: object = part.get("image_url", {}) upload_url: str if isinstance(image_url, str): @@ -322,16 +317,13 @@ class GigaChatConfig(BaseConfig): headers: Mapping[str, object], ) -> dict: # mutable-ok: request payload sent to httpx """Transform OpenAI request to GigaChat format.""" - # Transform messages giga_messages: Final = self._transform_messages(messages) - # Build request request_data: Final[dict[str, object]] = { "model": model.replace("gigachat/", ""), "messages": giga_messages, } - # Add optional params for key in [ "temperature", "top_p", @@ -343,7 +335,6 @@ class GigaChatConfig(BaseConfig): if key in optional_params: request_data[key] = optional_params[key] - # Add functions if present if "functions" in optional_params: request_data["functions"] = optional_params["functions"] if "function_call" in optional_params: @@ -358,10 +349,8 @@ class GigaChatConfig(BaseConfig): for i, msg in enumerate(messages): message = dict(msg) - # Remove unsupported fields message.pop("name", None) - # Transform roles role = message.get("role", "user") if role == "developer": message["role"] = "system" @@ -374,18 +363,15 @@ class GigaChatConfig(BaseConfig): if not isinstance(content, str) or not is_valid_json(content): message["content"] = json.dumps(content, ensure_ascii=False) - # Handle None content if message.get("content") is None: message["content"] = "" - # Handle list content (multimodal) - extract text and images content = message.get("content") if isinstance(content, list): message["content"], attachments = self._transform_list_content(content) if attachments: message["attachments"] = attachments - # Transform tool_calls to function_call tool_calls = message.get("tool_calls") if tool_calls and isinstance(tool_calls, list) and len(tool_calls) > 0: tool_call = tool_calls[0] @@ -436,13 +422,11 @@ class GigaChatConfig(BaseConfig): message_data = choice.get("message", {}) finish_reason = choice.get("finish_reason", "stop") - # Transform function_call to tool_calls or content if finish_reason == "function_call" and message_data.get("function_call"): func_call = message_data["function_call"] args = func_call.get("arguments", {}) if is_structured_output: - # Convert to content for structured output if isinstance(args, dict): content = json.dumps(args, ensure_ascii=False) else: @@ -452,7 +436,6 @@ class GigaChatConfig(BaseConfig): message_data.pop("functions_state_id", None) finish_reason = "stop" else: - # Convert to tool_calls format if isinstance(args, dict): args = json.dumps(args, ensure_ascii=False) message_data["tool_calls"] = [ @@ -468,7 +451,6 @@ class GigaChatConfig(BaseConfig): message_data.pop("function_call", None) finish_reason = "tool_calls" - # Clean up GigaChat-specific fields message_data.pop("functions_state_id", None) choices.append( diff --git a/litellm/llms/gigachat/embedding/transformation.py b/litellm/llms/gigachat/embedding/transformation.py index 2ec8324e33c..0db4475be8f 100644 --- a/litellm/llms/gigachat/embedding/transformation.py +++ b/litellm/llms/gigachat/embedding/transformation.py @@ -112,18 +112,10 @@ class GigaChatEmbeddingConfig(BaseEmbeddingConfig): "input": ["text1", "text2", ...] } """ - # Normalize input to list - if isinstance(input, str): - input_list: list = [input] # rebind-ok: locally scoped conversion - else: - input_list = input - - # Remove gigachat/ prefix from model if present - model = model.removeprefix("gigachat/") # rebind-ok: parameter reassignment for normalization - + normalized_input: Final = [input] if isinstance(input, str) else input # mutable-ok: preserve list API return { - "model": model, - "input": input_list, + "model": model.removeprefix("gigachat/"), + "input": normalized_input, } def transform_embedding_response( diff --git a/litellm/llms/gigachat/passthrough/transformation.py b/litellm/llms/gigachat/passthrough/transformation.py index a0edc6f5682..e1f73d04275 100644 --- a/litellm/llms/gigachat/passthrough/transformation.py +++ b/litellm/llms/gigachat/passthrough/transformation.py @@ -60,7 +60,6 @@ class GigaChatPassthroughConfig(BasePassthroughConfig): """ Set up headers with OAuth token. """ - # Get access token access_token: Final = get_access_token(credentials=api_key, litellm_params=litellm_params) headers["Authorization"] = f"Bearer {access_token}" # rebind-ok: mutating for OAuth setup @@ -82,7 +81,6 @@ class GigaChatPassthroughConfig(BasePassthroughConfig): from litellm.types.utils import LlmProviders, ModelResponse from litellm.utils import ProviderConfigManager - # cost tracking only for completions and embeddings if "completions" in endpoint: provider_chat_config: Final = ProviderConfigManager.get_provider_chat_config( provider=LlmProviders(custom_llm_provider), diff --git a/litellm/llms/gigachat/utils.py b/litellm/llms/gigachat/utils.py index cbb35cd1b57..ce7e848ed7f 100644 --- a/litellm/llms/gigachat/utils.py +++ b/litellm/llms/gigachat/utils.py @@ -4,7 +4,6 @@ from typing import Final from litellm.secret_managers.main import get_secret_str from litellm.types.utils import PromptTokensDetailsWrapper, Usage -# GigaChat API endpoint GIGACHAT_BASE_URL: Final = "https://gigachat.devices.sberbank.ru/api/v1" diff --git a/litellm/passthrough/main.py b/litellm/passthrough/main.py index 9095cee15a9..689c34b7a88 100644 --- a/litellm/passthrough/main.py +++ b/litellm/passthrough/main.py @@ -113,10 +113,8 @@ class AsyncPassthroughStreamingResponse(AsyncGenerator[Any, Any]): ) ) - # Compliant: Save a strong reference to prevent GC self._background_tasks.add(task) - # Remove the task from the set when it finishes to avoid memory leaks task.add_done_callback(self._background_tasks.discard) except Exception as e: # noqa: BLE001 # Safe catch-all for verbose logging verbose_logger.exception( @@ -578,7 +576,6 @@ def llm_passthrough_route( else: return response except Exception as e: - # provider_config is guaranteed non-None here due to the earlier guard assert provider_config is not None raise base_llm_http_handler._handle_error( e=e, diff --git a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py index 78d8ce296b8..b48b8d81494 100644 --- a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py +++ b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py @@ -1731,7 +1731,7 @@ def get_vertex_ai_allowed_incoming_headers(request: Request) -> dict: def get_vertex_pass_through_handler( - call_type: Literal["discovery", "aiplatform"], # noqa: UP037 + call_type: Literal["discovery", "aiplatform"], # noqa: UP037 # ruff reports quoted Literal values here ) -> BaseVertexAIPassThroughHandler: if call_type == "discovery": return VertexAIDiscoveryPassThroughHandler() @@ -2961,7 +2961,6 @@ async def handle_gigachat_passthrough_router_model( """ from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing - # Detect streaming based on request body is_streaming: Final = request_body.get("stream", False) # pyright: ignore[reportUnknownVariableType] # request_body is dict[Unknown, Unknown] data: dict[str, Any] = await _read_request_body( @@ -2997,7 +2996,6 @@ async def handle_gigachat_passthrough_router_model( data["json"] = request_body data["custom_llm_provider"] = "gigachat" - # Remove sensitive keys from data keys: Final = [ # mutable-ok: list of keys to remove from data "gigachat_auth_url", "gigachat_access_token", diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index 1ec8be88c9b..93803ce1005 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -10268,7 +10268,8 @@ class TestContextWindowEscalation: litellm_router_instance=_windowed_router(_SMALL, _BIG), complexity_router_config=_tier_config(session_affinity=True), ) - session_kwargs = lambda: {"metadata": {"session_id": "s-1", "user_api_key_hash": "k-1"}} # noqa: E731 + def session_kwargs() -> dict[str, object]: + return {"metadata": {"session_id": "s-1", "user_api_key_hash": "k-1"}} first = await router.async_pre_routing_hook( model="test-router", request_kwargs=session_kwargs(), messages=_OVERSIZED_TURNS @@ -10291,7 +10292,8 @@ class TestContextWindowEscalation: litellm_router_instance=_windowed_router(_SMALL, _BIG), complexity_router_config=_tier_config(session_affinity=True), ) - session_kwargs = lambda: {"metadata": {"session_id": "s-2", "user_api_key_hash": "k-2"}} # noqa: E731 + def session_kwargs() -> dict[str, object]: + return {"metadata": {"session_id": "s-2", "user_api_key_hash": "k-2"}} pinned = await router.async_pre_routing_hook( model="test-router", request_kwargs=session_kwargs(), messages=[{"role": "user", "content": "ok continue"}] diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 83c49afb538..f65ebd24599 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -1,12 +1,12 @@ { "LIT001": { - "limit": 22403 + "limit": 22402 }, "LIT002": { "limit": 26780 }, "LIT003": { - "limit": 269 + "limit": 268 }, "LIT004": { "limit": 40 @@ -27,10 +27,10 @@ "limit": 0 }, "LIT010": { - "limit": 16512 + "limit": 16511 }, "LIT011": { - "limit": 5537 + "limit": 5535 }, "LIT012": { "limit": 4495 From ab1161344199539bc8dec51161fb59d6d0acf703 Mon Sep 17 00:00:00 2001 From: milan Date: Wed, 5 Aug 2026 18:01:29 +0000 Subject: [PATCH 09/31] fix(bedrock): strip client_metadata from converse additionalModelRequestFields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../bedrock/chat/converse_transformation.py | 1 + .../chat/test_converse_transformation.py | 24 +++++++++++++++++++ 2 files changed, 25 insertions(+) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 395d99a4caa..9e40cb5ee5b 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -1324,6 +1324,7 @@ class AmazonConverseConfig(BaseConfig): ) additional_request_params.pop("parallel_tool_calls", None) + additional_request_params.pop("client_metadata", None) # Only set the topK value in for models that support it additional_request_params.update(self._handle_top_k_value(model, inference_params, drop_params)) diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index 63f895e1819..bd68857d664 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -979,6 +979,30 @@ def test_config_blocks_do_not_leak_into_inference_config(): assert data["serviceTier"] == {"type": "priority"} +def test_client_metadata_stripped_from_converse_request(): + """``client_metadata`` sent by codex must not reach Bedrock as a passthrough model field. + + Converse forwards ``additionalModelRequestFields`` verbatim to the model, and Anthropic + rejects the request with "client_metadata: Extra inputs are not permitted". + """ + config = AmazonConverseConfig() + + data = config._transform_request_helper( + model="anthropic.claude-opus-4-8", + system_content_blocks=[], + optional_params={ + "maxTokens": 16, + "anthropic_beta": ["computer-use-2025-01-24"], + "client_metadata": {"originator": "codex_cli_rs"}, + }, + messages=None, + ) + + fields = data.get("additionalModelRequestFields", {}) + assert "client_metadata" not in fields + assert fields["anthropic_beta"] == ["computer-use-2025-01-24"] + + def test_parallel_tool_calls_config_kept_for_sonnet_5(monkeypatch): old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") old_cost = litellm.model_cost From 2063c29f5d95f8dd00eef3fd7dfcbc1df05787b8 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 18:45:26 +0000 Subject: [PATCH 10/31] fix(anthropic): upgrade legacy thinking to adaptive on adaptive-only models for chat and Bedrock Converse Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/anthropic/chat/transformation.py | 3 + litellm/llms/anthropic/common_utils.py | 47 +++++++++++- .../messages/transformation.py | 47 +----------- .../bedrock/chat/converse_transformation.py | 3 + .../test_anthropic_chat_transformation.py | 25 ++++++ .../chat/test_converse_transformation.py | 76 +++++++++++++++++++ 6 files changed, 154 insertions(+), 47 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index e1387a9068c..319eecfac2c 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1544,6 +1544,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): optional_params.pop("thinking", None) else: optional_params["thinking"] = value + AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model( + model=model, optional_params=optional_params, custom_llm_provider=self._resolved_provider + ) elif param == "reasoning_effort": # Accept both string ("low") and dict ({"effort": "low", # "summary": "concise"}). The Responses->Chat parser keeps the diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index 9871001bf66..ede93c6deb2 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -13,7 +13,12 @@ import httpx from pydantic import BaseModel, ConfigDict, TypeAdapter, ValidationError import litellm -from litellm.constants import DEFAULT_MODEL_CREATED_AT_TIME +from litellm.constants import ( + DEFAULT_MODEL_CREATED_AT_TIME, + DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, +) from litellm.litellm_core_utils.prompt_templates.common_utils import ( get_file_ids_from_messages, ) @@ -490,6 +495,46 @@ class AnthropicModelInfo(BaseLLMModelInfo): ) optional_params.pop("thinking", None) + @staticmethod + def translate_legacy_thinking_for_adaptive_model( + model: str, + optional_params: MutableMapping[str, object], # mutable-ok: in-place out-param, as in maybe_drop_disabled_thinking + custom_llm_provider: str, + ) -> None: + """Translate legacy ``thinking.type=enabled`` to adaptive for the + adaptive-thinking models that reject it (4.7+ and the 5 families). + Models flagged ``supports_legacy_thinking`` (the 4.6 family) accept the + legacy shape natively, so it is forwarded verbatim and the caller's + ``budget_tokens`` cap keeps applying. Caller-provided + ``output_config.effort`` is never overridden. + """ + if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider): + return + if AnthropicModelInfo._supports_legacy_thinking(model, custom_llm_provider): + return + thinking: Final = optional_params.get("thinking") + if not isinstance(thinking, dict) or thinking.get("type") != "enabled": + return + + budget: Final = int(thinking.get("budget_tokens") or 0) + if budget >= DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET and ( + AnthropicModelInfo._supports_model_capability(model, "supports_xhigh_reasoning_effort", custom_llm_provider) + ): + effort = "xhigh" + elif budget >= DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET: + effort = "high" + elif budget >= DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET: + effort = "medium" + else: + effort = "low" + + optional_params["thinking"] = {"type": "adaptive"} + existing_output_config = optional_params.get("output_config") + if not isinstance(existing_output_config, dict): + existing_output_config = {} + existing_output_config.setdefault("effort", effort) + optional_params["output_config"] = existing_output_config + def is_effort_used( self, optional_params: dict | None, diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index 3d62b8b4784..988f81c9eb4 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -3,11 +3,6 @@ from typing import Any, Final import httpx -from litellm.constants import ( - DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, - DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, - DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, -) from litellm.exceptions import AuthenticationError from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.litellm_logging import verbose_logger @@ -400,46 +395,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): existing_output_config.setdefault("effort", mapped_effort) optional_params["output_config"] = existing_output_config - @staticmethod - def _translate_legacy_thinking_for_adaptive_model( - model: str, optional_params: dict, custom_llm_provider: str - ) -> None: - """Translate legacy ``thinking.type=enabled`` to adaptive for the - adaptive-thinking models that reject it (4.7+ and the 5 families). - Models flagged ``supports_legacy_thinking`` (the 4.6 family) accept the - legacy shape natively, so it is forwarded verbatim and the caller's - ``budget_tokens`` cap keeps applying. Caller-provided - ``output_config.effort`` is never overridden. - """ - from litellm.llms.anthropic.chat.transformation import AnthropicConfig - - if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider): - return - if AnthropicModelInfo._supports_legacy_thinking(model, custom_llm_provider): - return - thinking: Final = optional_params.get("thinking") - if not isinstance(thinking, dict) or thinking.get("type") != "enabled": - return - - budget: Final = int(thinking.get("budget_tokens") or 0) - if budget >= DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET and ( - AnthropicConfig._supports_effort_level(model, "xhigh", custom_llm_provider) - ): - effort = "xhigh" - elif budget >= DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET: - effort = "high" - elif budget >= DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET: - effort = "medium" - else: - effort = "low" - - optional_params["thinking"] = {"type": "adaptive"} - existing_output_config = optional_params.get("output_config") - if not isinstance(existing_output_config, dict): - existing_output_config = {} - existing_output_config.setdefault("effort", effort) - optional_params["output_config"] = existing_output_config - @staticmethod def _translate_adaptive_effort_for_non_adaptive_model( model: str, optional_params: dict, max_tokens: int | None, custom_llm_provider: str @@ -606,7 +561,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): custom_llm_provider=self._resolved_provider, ) - self._translate_legacy_thinking_for_adaptive_model( + AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model( model=model, optional_params=anthropic_messages_optional_request_params, custom_llm_provider=self._resolved_provider, diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 395d99a4caa..0a378dfc11c 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -934,6 +934,9 @@ class AmazonConverseConfig(BaseConfig): litellm.verbose_logger.warning(DROP_UNSUPPORTED_ADAPTIVE_THINKING_WARNING, model) else: optional_params["thinking"] = value + AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model( + model=model, optional_params=optional_params, custom_llm_provider="bedrock" + ) elif param == "reasoning_effort" and isinstance(value, str): self._handle_reasoning_effort_parameter( model=model, reasoning_effort=value, optional_params=optional_params diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 25e2c3cda80..4f30e7d10f0 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -3127,6 +3127,31 @@ def test_reasoning_effort_accepts_dict_shape_for_non_adaptive_model( ) +@pytest.mark.parametrize( + "model,budget_tokens,expected", + [ + ("claude-opus-4-8", 4096, ({"type": "adaptive"}, {"effort": "high"})), + ("claude-opus-4-7", 24000, ({"type": "adaptive"}, {"effort": "xhigh"})), + ("claude-opus-4-6", 4096, ({"type": "enabled", "budget_tokens": 4096}, None)), + ("claude-sonnet-4-5-20250929", 4096, ({"type": "enabled", "budget_tokens": 4096}, None)), + ], +) +def test_legacy_thinking_translated_to_adaptive_on_adaptive_only_models(model, budget_tokens, expected): + """Adaptive-only models reject thinking={type: enabled} with a 400, so the + legacy shape must be upgraded to adaptive + output_config.effort on + /chat/completions too, while models that accept it keep the caller's budget.""" + config = AnthropicConfig() + + result = config.map_openai_params( + non_default_params={"thinking": {"type": "enabled", "budget_tokens": budget_tokens}, "max_tokens": 64000}, + optional_params={}, + model=model, + drop_params=False, + ) + + assert (result["thinking"], result.get("output_config")) == expected + + @pytest.mark.parametrize( "bad_value", [ diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index 63f895e1819..b729c9366cc 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -6269,6 +6269,82 @@ def test_adaptive_thinking_passes_through_on_46_plus_converse(model): assert optional_params.get("thinking") == {"type": "adaptive"} +@pytest.mark.parametrize( + "model,budget_tokens,expected_effort", + [ + ("anthropic.claude-opus-4-8", 4096, "high"), + ("us.anthropic.claude-opus-4-8", 2000, "low"), + ("global.anthropic.claude-opus-4-8", 12000, "xhigh"), + ("us.anthropic.claude-opus-4-7", 3000, "medium"), + ("anthropic.claude-fable-5", 4096, "high"), + ], +) +def test_legacy_thinking_translated_to_adaptive_on_adaptive_only_converse(model, budget_tokens, expected_effort): + """Adaptive-only models (4.7+, 5 families) reject thinking={type: enabled} + with a 400 on Bedrock Converse, so the legacy shape from callers like Claude + Code must be upgraded to thinking={type: adaptive} + output_config.effort + derived from budget_tokens, matching the /v1/messages passthrough.""" + config = AmazonConverseConfig() + + optional_params = config.map_openai_params( + non_default_params={"thinking": {"type": "enabled", "budget_tokens": budget_tokens}, "max_tokens": 64000}, + optional_params={}, + model=model, + drop_params=False, + ) + request = config.transform_request( + model=model, + messages=[{"role": "user", "content": "hi"}], + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert request["additionalModelRequestFields"]["thinking"] == {"type": "adaptive"} + assert request["additionalModelRequestFields"]["output_config"] == {"effort": expected_effort} + + +def test_legacy_thinking_translation_keeps_caller_output_config_effort_converse(): + config = AmazonConverseConfig() + + optional_params = config.map_openai_params( + non_default_params={ + "output_config": {"effort": "low"}, + "thinking": {"type": "enabled", "budget_tokens": 12000}, + "max_tokens": 64000, + }, + optional_params={}, + model="anthropic.claude-opus-4-8", + drop_params=False, + ) + + assert optional_params["thinking"] == {"type": "adaptive"} + assert optional_params["output_config"] == {"effort": "low"} + + +@pytest.mark.parametrize( + "model", + [ + "us.anthropic.claude-opus-4-6", + "anthropic.claude-3-5-sonnet-20241022-v2:0", + ], +) +def test_legacy_thinking_forwarded_verbatim_when_model_accepts_it_converse(model): + """The 4.6 family and pre-adaptive models accept thinking={type: enabled} + natively, so the caller's budget_tokens cap must keep applying.""" + config = AmazonConverseConfig() + + optional_params = config.map_openai_params( + non_default_params={"thinking": {"type": "enabled", "budget_tokens": 4096}, "max_tokens": 8192}, + optional_params={}, + model=model, + drop_params=False, + ) + + assert optional_params["thinking"] == {"type": "enabled", "budget_tokens": 4096} + assert "output_config" not in optional_params + + def test_adaptive_thinking_dropped_when_max_tokens_too_small_converse(): """When max_tokens can't fit even the minimum thinking budget, the raw adaptive block must be dropped entirely rather than translated, so the From b8dd27a77fdaaa390761a99bf27737a255c37e3a Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 19:00:46 +0000 Subject: [PATCH 11/31] style(anthropic): keep mutable-ok annotation within ruff format width Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/anthropic/common_utils.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index ede93c6deb2..f2fcbc6c232 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -498,7 +498,7 @@ class AnthropicModelInfo(BaseLLMModelInfo): @staticmethod def translate_legacy_thinking_for_adaptive_model( model: str, - optional_params: MutableMapping[str, object], # mutable-ok: in-place out-param, as in maybe_drop_disabled_thinking + optional_params: MutableMapping[str, object], # mutable-ok: in-place out-param like the sibling helpers custom_llm_provider: str, ) -> None: """Translate legacy ``thinking.type=enabled`` to adaptive for the From ed4343a02645e19590657ae257e6ae43b95f8e48 Mon Sep 17 00:00:00 2001 From: milan Date: Tue, 1 Sep 2026 19:04:32 +0000 Subject: [PATCH 12/31] fix(bedrock): scope client_metadata drop to anthropic converse models Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../bedrock/chat/converse_transformation.py | 8 ++++++- .../chat/test_converse_transformation.py | 24 ++++++++++++++++--- 2 files changed, 28 insertions(+), 4 deletions(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 9e40cb5ee5b..8d1d905a129 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -1324,7 +1324,13 @@ class AmazonConverseConfig(BaseConfig): ) additional_request_params.pop("parallel_tool_calls", None) - additional_request_params.pop("client_metadata", None) + + if base_model.startswith("anthropic") and additional_request_params.pop("client_metadata", None) is not None: + litellm.verbose_logger.debug( + "Bedrock Converse: dropping `client_metadata` for model=%s, Anthropic rejects it with " + "'client_metadata: Extra inputs are not permitted'", + model, + ) # Only set the topK value in for models that support it additional_request_params.update(self._handle_top_k_value(model, inference_params, drop_params)) diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index bd68857d664..3600e366b5a 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -979,8 +979,9 @@ def test_config_blocks_do_not_leak_into_inference_config(): assert data["serviceTier"] == {"type": "priority"} -def test_client_metadata_stripped_from_converse_request(): - """``client_metadata`` sent by codex must not reach Bedrock as a passthrough model field. +@pytest.mark.parametrize("model", ["anthropic.claude-opus-4-8", "us.anthropic.claude-opus-4-8"]) +def test_client_metadata_stripped_for_anthropic_converse_request(model): + """``client_metadata`` sent by codex must not reach Anthropic as a passthrough model field. Converse forwards ``additionalModelRequestFields`` verbatim to the model, and Anthropic rejects the request with "client_metadata: Extra inputs are not permitted". @@ -988,7 +989,7 @@ def test_client_metadata_stripped_from_converse_request(): config = AmazonConverseConfig() data = config._transform_request_helper( - model="anthropic.claude-opus-4-8", + model=model, system_content_blocks=[], optional_params={ "maxTokens": 16, @@ -1003,6 +1004,23 @@ def test_client_metadata_stripped_from_converse_request(): assert fields["anthropic_beta"] == ["computer-use-2025-01-24"] +def test_client_metadata_kept_for_non_anthropic_converse_request(): + """Only Anthropic is known to reject ``client_metadata``, so other families keep the passthrough.""" + config = AmazonConverseConfig() + + data = config._transform_request_helper( + model="amazon.nova-pro-v1:0", + system_content_blocks=[], + optional_params={ + "maxTokens": 16, + "client_metadata": {"originator": "codex_cli_rs"}, + }, + messages=None, + ) + + assert data["additionalModelRequestFields"]["client_metadata"] == {"originator": "codex_cli_rs"} + + def test_parallel_tool_calls_config_kept_for_sonnet_5(monkeypatch): old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") old_cost = litellm.model_cost From b96121efb173965405aa521ffbbba026ce73d3a4 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 19:10:22 +0000 Subject: [PATCH 13/31] refactor(anthropic): build adaptive output_config in one shot in legacy thinking helper Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/anthropic/common_utils.py | 37 +++++++++++++++----------- 1 file changed, 21 insertions(+), 16 deletions(-) diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index f2fcbc6c232..17fa8022388 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -516,24 +516,29 @@ class AnthropicModelInfo(BaseLLMModelInfo): if not isinstance(thinking, dict) or thinking.get("type") != "enabled": return - budget: Final = int(thinking.get("budget_tokens") or 0) - if budget >= DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET and ( + effort: Final = AnthropicModelInfo._legacy_budget_to_effort( + model=model, + budget_tokens=int(thinking.get("budget_tokens") or 0), + custom_llm_provider=custom_llm_provider, + ) + existing_output_config: Final = optional_params.get("output_config") + optional_params["thinking"] = {"type": "adaptive"} + optional_params["output_config"] = { + "effort": effort, + **(existing_output_config if isinstance(existing_output_config, dict) else MappingProxyType({})), + } + + @staticmethod + def _legacy_budget_to_effort(model: str, budget_tokens: int, custom_llm_provider: str) -> str: + if budget_tokens >= DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET and ( AnthropicModelInfo._supports_model_capability(model, "supports_xhigh_reasoning_effort", custom_llm_provider) ): - effort = "xhigh" - elif budget >= DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET: - effort = "high" - elif budget >= DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET: - effort = "medium" - else: - effort = "low" - - optional_params["thinking"] = {"type": "adaptive"} - existing_output_config = optional_params.get("output_config") - if not isinstance(existing_output_config, dict): - existing_output_config = {} - existing_output_config.setdefault("effort", effort) - optional_params["output_config"] = existing_output_config + return "xhigh" + if budget_tokens >= DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET: + return "high" + if budget_tokens >= DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET: + return "medium" + return "low" def is_effort_used( self, From bba951c5ebe20871caab9848c474585c91fa3535 Mon Sep 17 00:00:00 2001 From: mateo Date: Tue, 1 Sep 2026 19:26:19 +0000 Subject: [PATCH 14/31] fix(bedrock): drop client_metadata for ARNs that hide the model family Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../bedrock/chat/converse_transformation.py | 4 +- litellm/llms/bedrock/common_utils.py | 9 ++++ .../chat/test_converse_transformation.py | 43 +++++++++++++++++++ 3 files changed, 55 insertions(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 8d1d905a129..99d45bfc94b 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -86,6 +86,7 @@ from litellm.utils import ( from ..common_utils import ( BedrockError, BedrockModelInfo, + bedrock_arn_hides_model_family, bedrock_converse_supports_parallel_tool_use_config, get_anthropic_beta_from_headers, get_bedrock_tool_name, @@ -1325,7 +1326,8 @@ class AmazonConverseConfig(BaseConfig): additional_request_params.pop("parallel_tool_calls", None) - if base_model.startswith("anthropic") and additional_request_params.pop("client_metadata", None) is not None: + drops_client_metadata: Final = base_model.startswith("anthropic") or bedrock_arn_hides_model_family(model) + if drops_client_metadata and additional_request_params.pop("client_metadata", None) is not None: litellm.verbose_logger.debug( "Bedrock Converse: dropping `client_metadata` for model=%s, Anthropic rejects it with " "'client_metadata: Extra inputs are not permitted'", diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 9cbceb4880c..3b82d98ceee 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -720,6 +720,15 @@ def get_bedrock_base_model(model: str) -> str: return model +def bedrock_arn_hides_model_family(model: str) -> bool: + """ + True for an ARN-addressed model whose base name carries no ``provider.model`` + id, such as an application inference profile or a provisioned throughput ARN. + Callers that gate behavior on the model family cannot resolve one here. + """ + return "arn:" in model.lower() and "." not in get_bedrock_base_model(model) + + def bedrock_converse_supports_parallel_tool_use_config(model: str) -> bool: return any( (litellm.model_cost.get(candidate) or {}).get("supports_parallel_tool_use_config") is True diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index 3600e366b5a..423e3a5a7c3 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -1021,6 +1021,49 @@ def test_client_metadata_kept_for_non_anthropic_converse_request(): assert data["additionalModelRequestFields"]["client_metadata"] == {"originator": "codex_cli_rs"} +@pytest.mark.parametrize( + "model", + [ + "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abcdef123456", + "arn:aws:bedrock:us-east-1:123456789012:provisioned-model/abcdef123456", + ], +) +def test_client_metadata_stripped_for_arn_models_converse(model): + """An ARN hides which family serves the request, and pointing one at Claude is how + teams route codex traffic, so the field has to go there too or the 400 comes back.""" + config = AmazonConverseConfig() + + data = config._transform_request_helper( + model=model, + system_content_blocks=[], + optional_params={ + "maxTokens": 16, + "client_metadata": {"originator": "codex_cli_rs"}, + }, + messages=None, + ) + + assert "client_metadata" not in data.get("additionalModelRequestFields", {}) + + +def test_client_metadata_kept_for_arn_naming_a_non_anthropic_family(): + """An inference profile ARN that still spells out the family is resolvable, so a + non-Anthropic one keeps its passthrough.""" + config = AmazonConverseConfig() + + data = config._transform_request_helper( + model="arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.amazon.nova-pro-v1:0", + system_content_blocks=[], + optional_params={ + "maxTokens": 16, + "client_metadata": {"originator": "codex_cli_rs"}, + }, + messages=None, + ) + + assert data["additionalModelRequestFields"]["client_metadata"] == {"originator": "codex_cli_rs"} + + def test_parallel_tool_calls_config_kept_for_sonnet_5(monkeypatch): old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") old_cost = litellm.model_cost From 1400070d711f645290fb382564e1f21200c5e610 Mon Sep 17 00:00:00 2001 From: mateo Date: Wed, 2 Sep 2026 07:58:09 +0000 Subject: [PATCH 15/31] chore(techdebt): clear fresh debt from the 2026-09-01 window Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- basedpyright-code-budget.json | 6 +++--- .../integrations/SlackAlerting/slack_alerting.py | 9 ++++----- .../websearch_interception/handler.py | 1 - .../_experimental/mcp_server/rest_endpoints.py | 13 +++++++++---- .../guardrails/guardrail_hooks/alice/alice.py | 16 ++++++++-------- type-discipline-budget.json | 6 +++--- 6 files changed, 27 insertions(+), 24 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index 2d4adc02234..f14f8e002dd 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5601 }, "reportMissingTypeArgument": { - "limit": 15300 + "limit": 15298 }, "reportMissingTypeStubs": { "limit": 40 @@ -105,13 +105,13 @@ "limit": 109 }, "reportUnknownMemberType": { - "limit": 38347 + "limit": 38344 }, "reportUnknownParameterType": { "limit": 19626 }, "reportUnknownVariableType": { - "limit": 29884 + "limit": 29880 }, "reportUnnecessaryCast": { "limit": 111 diff --git a/litellm/integrations/SlackAlerting/slack_alerting.py b/litellm/integrations/SlackAlerting/slack_alerting.py index 748ef938cea..dc41c7dadc8 100644 --- a/litellm/integrations/SlackAlerting/slack_alerting.py +++ b/litellm/integrations/SlackAlerting/slack_alerting.py @@ -1955,11 +1955,10 @@ Model Info: if not thresholds_enabled and not anomalies_enabled: return - if prisma_client is None: - from litellm.proxy.proxy_server import prisma_client as global_prisma_client + from litellm.proxy.proxy_server import prisma_client as global_prisma_client - prisma_client = global_prisma_client # rebind-ok: fall back to the proxy's global client - if prisma_client is None: + client: Final = prisma_client if prisma_client is not None else global_prisma_client + if client is None: return from litellm.integrations.SlackAlerting.user_spend_alerts import ( @@ -1970,7 +1969,7 @@ Model Info: try: today: Final = datetime.datetime.now(datetime.timezone.utc).date() rows: Final = await fetch_user_spend_rows( - prisma_client=prisma_client, + prisma_client=client, today=today, baseline_days=self.alerting_args.spend_anomaly_baseline_days, ) diff --git a/litellm/integrations/websearch_interception/handler.py b/litellm/integrations/websearch_interception/handler.py index dc61ee38a8c..2d737bc34e7 100644 --- a/litellm/integrations/websearch_interception/handler.py +++ b/litellm/integrations/websearch_interception/handler.py @@ -419,7 +419,6 @@ class WebSearchInterceptionLogger(CustomLogger): if call_type in (CallTypes.responses, CallTypes.aresponses): return self._convert_responses_tools(kwargs=kwargs, tools=tools) - # Check if any tool is a web search tool (native or already LiteLLM standard) has_websearch: Final = any(is_web_search_tool(t) for t in tools) if not has_websearch: diff --git a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py index d1ef73a15cd..90474bfc5e6 100644 --- a/litellm/proxy/_experimental/mcp_server/rest_endpoints.py +++ b/litellm/proxy/_experimental/mcp_server/rest_endpoints.py @@ -92,6 +92,9 @@ def _connection_error_message(exc: BaseException) -> str: if MCP_AVAILABLE: + from mcp.types import Tool as MCPTool + + from litellm.experimental_mcp_client.client import MCPClient from litellm.proxy._experimental.mcp_server.mcp_server_manager import ( _UPSTREAM_OAUTH_DISCOVERY_AUTH_TYPES, global_mcp_server_manager, @@ -876,7 +879,6 @@ if MCP_AVAILABLE: return (), classify_list_exception(e) return tools_result, ServerListOk(tool_count=len(tools_result)) - # Query all servers the user has access to queried_servers: Final = tuple( server for server in map(global_mcp_server_manager.get_mcp_server_by_id, allowed_server_ids) @@ -1141,6 +1143,11 @@ if MCP_AVAILABLE: scopes: Final[list[str] | None] = scopes_raw if isinstance(scopes_raw, list) else None return client_id, client_secret, scopes + async def _list_tools_within(client: MCPClient, deadline: float) -> list[MCPTool] | None: + with anyio.move_on_after(deadline): + return await client.list_tools(raise_on_error=True) + return None + async def _execute_with_mcp_client( request: NewMCPServerRequest, operation: Callable[..., Awaitable[Mapping[str, object]]], @@ -1422,9 +1429,7 @@ if MCP_AVAILABLE: getattr(client, "timeout", MCP_CLIENT_TIMEOUT) or MCP_CLIENT_TIMEOUT, MCP_TOOL_LISTING_TIMEOUT, ) - list_tools_result = None # rebind-ok: set inside the timeout scope below - with anyio.move_on_after(listing_deadline): - list_tools_result = await client.list_tools(raise_on_error=True) # rebind-ok: fills the init above + list_tools_result: Final = await _list_tools_within(client, listing_deadline) if list_tools_result is None: verbose_logger.warning( "MCP tools/list preview timed out after %s seconds while paginating upstream tools", diff --git a/litellm/proxy/guardrails/guardrail_hooks/alice/alice.py b/litellm/proxy/guardrails/guardrail_hooks/alice/alice.py index 27018769909..9cabac2d0fa 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/alice/alice.py +++ b/litellm/proxy/guardrails/guardrail_hooks/alice/alice.py @@ -8,6 +8,7 @@ import json import os from collections.abc import Mapping +from itertools import islice from typing import ( TYPE_CHECKING, Any, # noqa: TID251 # **kwargs forwards verbatim to CustomGuardrail.__init__; see ruff-strict.toml @@ -341,19 +342,18 @@ def _json_safe( if depth >= _MAX_DEPTH or id(value) in seen: return None - nested: Final = seen | {id(value)} # mutable-ok: one-shot set literal, unioned into a frozenset immediately + nested: Final = seen | frozenset((id(value),)) if isinstance(value, dict): - out: dict[str, object] = {} # mutable-ok: bounded accumulator local to this call, never escapes as-is - for key, item in list(value.items())[:_MAX_ITEMS]: # mutable-ok: list() only to slice an unordered view - if isinstance(key, str) and key not in strip_keys: - out[key] = _json_safe(item, depth + 1, nested, strip_keys) - return out + return { + key: _json_safe(item, depth + 1, nested, strip_keys) + for key, item in islice(value.items(), _MAX_ITEMS) + if isinstance(key, str) and key not in strip_keys + } if isinstance(value, (list, tuple, set, frozenset)): return [ # mutable-ok: return value is a one-shot list, discarded by the caller after use - _json_safe(item, depth + 1, nested, strip_keys) - for item in list(value)[:_MAX_ITEMS] # mutable-ok: list() only to slice an unordered view + _json_safe(item, depth + 1, nested, strip_keys) for item in islice(value, _MAX_ITEMS) ] dump: Final = getattr(value, "model_dump", None) diff --git a/type-discipline-budget.json b/type-discipline-budget.json index fbf998533f8..0f39b32670a 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -6,7 +6,7 @@ "limit": 26777 }, "LIT003": { - "limit": 266 + "limit": 265 }, "LIT004": { "limit": 40 @@ -27,10 +27,10 @@ "limit": 0 }, "LIT010": { - "limit": 16504 + "limit": 16502 }, "LIT011": { - "limit": 5531 + "limit": 5529 }, "LIT012": { "limit": 4495 From dba190842cea106ab4b03880861038ddbe6aae42 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:26:59 -0700 Subject: [PATCH 16/31] fix(anthropic): keep the cache_control normalizer inside the type-discipline budget --- litellm/llms/anthropic/common_utils.py | 38 +++++++++++++++++--------- 1 file changed, 25 insertions(+), 13 deletions(-) diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index b5bfb32c0c6..19d3d6d7043 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -1404,13 +1404,33 @@ def _with_portable_cache_control_in_message(message: object) -> object: return message return { # mutable-ok: JSON wire format **message, - "content": [_with_portable_cache_control_in_content_block(block) for block in content], + "content": [ # mutable-ok: JSON wire format + _with_portable_cache_control_in_content_block(block) for block in content + ], } -def normalize_cache_control_in_anthropic_payload( # mutable-ok: JSON wire format +def _with_portable_cache_control_in_messages(messages: object) -> object: + if isinstance(messages, str) or not isinstance(messages, Sequence): + return messages + return [ # mutable-ok: JSON wire format + _with_portable_cache_control_in_message(message) for message in messages + ] + + +def _with_portable_cache_control_in_scoped_value(key: str, value: object) -> object: + match key: + case "system" | "tools": + return _with_portable_cache_control_in_blocks(value) + case "messages": + return _with_portable_cache_control_in_messages(value) + case _: + return value + + +def normalize_cache_control_in_anthropic_payload( payload: Mapping[str, object], -) -> dict[str, object]: +) -> dict[str, object]: # mutable-ok: JSON wire format """ Return a copy of an Anthropic /v1/messages payload with every ``cache_control`` entry reduced to ``{"type": }`` @@ -1427,17 +1447,9 @@ def normalize_cache_control_in_anthropic_payload( # mutable-ok: JSON wire forma dropped entirely. The caller's payload is never mutated. """ portable: Final = _with_portable_cache_control(payload) - scoped: Final = { # mutable-ok: JSON wire format - key: ( - _with_portable_cache_control_in_blocks(value) - if key in ("system", "tools") - else [_with_portable_cache_control_in_message(message) for message in value] - if key == "messages" and isinstance(value, Sequence) and not isinstance(value, str) - else value - ) - for key, value in portable.items() + return { # mutable-ok: JSON wire format + key: _with_portable_cache_control_in_scoped_value(key, value) for key, value in portable.items() } - return scoped def process_anthropic_headers(headers: httpx.Headers | dict) -> dict: From 53da9bca8e45af86507cb6b5c83736290913ba71 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:29:44 -0700 Subject: [PATCH 17/31] fix(bedrock): drop client_metadata for every converse model --- .../bedrock/chat/converse_transformation.py | 10 +-- litellm/llms/bedrock/common_utils.py | 9 -- .../chat/test_converse_transformation.py | 85 +++---------------- 3 files changed, 15 insertions(+), 89 deletions(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 38b9569856a..df52b78f6b5 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -86,7 +86,6 @@ from litellm.utils import ( from ..common_utils import ( BedrockError, BedrockModelInfo, - bedrock_arn_hides_model_family, bedrock_converse_supports_parallel_tool_use_config, bedrock_model_accepts_cache_points, get_anthropic_beta_from_headers, @@ -1335,14 +1334,7 @@ class AmazonConverseConfig(BaseConfig): ) additional_request_params.pop("parallel_tool_calls", None) - - drops_client_metadata: Final = base_model.startswith("anthropic") or bedrock_arn_hides_model_family(model) - if drops_client_metadata and additional_request_params.pop("client_metadata", None) is not None: - litellm.verbose_logger.debug( - "Bedrock Converse: dropping `client_metadata` for model=%s, Anthropic rejects it with " - "'client_metadata: Extra inputs are not permitted'", - model, - ) + additional_request_params.pop("client_metadata", None) # Only set the topK value in for models that support it additional_request_params.update(self._handle_top_k_value(model, inference_params, drop_params)) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index cf18e3e3ec8..66ee5f10679 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -809,15 +809,6 @@ def get_bedrock_base_model(model: str) -> str: return model -def bedrock_arn_hides_model_family(model: str) -> bool: - """ - True for an ARN-addressed model whose base name carries no ``provider.model`` - id, such as an application inference profile or a provisioned throughput ARN. - Callers that gate behavior on the model family cannot resolve one here. - """ - return "arn:" in model.lower() and "." not in get_bedrock_base_model(model) - - def bedrock_converse_supports_parallel_tool_use_config(model: str) -> bool: return any( (litellm.model_cost.get(candidate) or {}).get("supports_parallel_tool_use_config") is True diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index 7556e1624be..fb165c38ef9 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -979,16 +979,19 @@ def test_config_blocks_do_not_leak_into_inference_config(): assert data["serviceTier"] == {"type": "priority"} -@pytest.mark.parametrize("model", ["anthropic.claude-opus-4-8", "us.anthropic.claude-opus-4-8"]) -def test_client_metadata_stripped_for_anthropic_converse_request(model): - """``client_metadata`` sent by codex must not reach Anthropic as a passthrough model field. - - Converse forwards ``additionalModelRequestFields`` verbatim to the model, and Anthropic - rejects the request with "client_metadata: Extra inputs are not permitted". - """ - config = AmazonConverseConfig() - - data = config._transform_request_helper( +@pytest.mark.parametrize( + "model", + [ + "anthropic.claude-opus-4-8", + "us.anthropic.claude-opus-4-8", + "amazon.nova-pro-v1:0", + "us.meta.llama4-maverick-17b-instruct-v1:0", + "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abcdef123456", + "arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.amazon.nova-pro-v1:0", + ], +) +def test_client_metadata_stripped_from_converse_request(model): + data = AmazonConverseConfig()._transform_request_helper( model=model, system_content_blocks=[], optional_params={ @@ -999,71 +1002,11 @@ def test_client_metadata_stripped_for_anthropic_converse_request(model): messages=None, ) - fields = data.get("additionalModelRequestFields", {}) + fields = data["additionalModelRequestFields"] assert "client_metadata" not in fields assert fields["anthropic_beta"] == ["computer-use-2025-01-24"] -def test_client_metadata_kept_for_non_anthropic_converse_request(): - """Only Anthropic is known to reject ``client_metadata``, so other families keep the passthrough.""" - config = AmazonConverseConfig() - - data = config._transform_request_helper( - model="amazon.nova-pro-v1:0", - system_content_blocks=[], - optional_params={ - "maxTokens": 16, - "client_metadata": {"originator": "codex_cli_rs"}, - }, - messages=None, - ) - - assert data["additionalModelRequestFields"]["client_metadata"] == {"originator": "codex_cli_rs"} - - -@pytest.mark.parametrize( - "model", - [ - "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abcdef123456", - "arn:aws:bedrock:us-east-1:123456789012:provisioned-model/abcdef123456", - ], -) -def test_client_metadata_stripped_for_arn_models_converse(model): - """An ARN hides which family serves the request, and pointing one at Claude is how - teams route codex traffic, so the field has to go there too or the 400 comes back.""" - config = AmazonConverseConfig() - - data = config._transform_request_helper( - model=model, - system_content_blocks=[], - optional_params={ - "maxTokens": 16, - "client_metadata": {"originator": "codex_cli_rs"}, - }, - messages=None, - ) - - assert "client_metadata" not in data.get("additionalModelRequestFields", {}) - - -def test_client_metadata_kept_for_arn_naming_a_non_anthropic_family(): - """An inference profile ARN that still spells out the family is resolvable, so a - non-Anthropic one keeps its passthrough.""" - config = AmazonConverseConfig() - - data = config._transform_request_helper( - model="arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.amazon.nova-pro-v1:0", - system_content_blocks=[], - optional_params={ - "maxTokens": 16, - "client_metadata": {"originator": "codex_cli_rs"}, - }, - messages=None, - ) - - assert data["additionalModelRequestFields"]["client_metadata"] == {"originator": "codex_cli_rs"} - - def test_parallel_tool_calls_config_kept_for_sonnet_5(monkeypatch): old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") old_cost = litellm.model_cost From dbc126cfc97734e47066ab988c3ac1090e7f830a Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:07:30 -0700 Subject: [PATCH 18/31] fix(hosted_vllm): forward truncate_prompt_tokens on rerank requests --- .../llms/hosted_vllm/rerank/transformation.py | 21 ++- litellm/types/rerank.py | 21 ++- .../test_hosted_vllm_rerank_transformation.py | 122 +++++++++++++++++- 3 files changed, 154 insertions(+), 10 deletions(-) diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index 0e8fa294f5d..265eb350fc6 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -3,6 +3,7 @@ Transformation logic for Hosted VLLM rerank """ from collections.abc import Mapping +from types import MappingProxyType from typing import Any, Final import httpx @@ -13,6 +14,7 @@ from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.base_llm.rerank.transformation import BaseRerankConfig from litellm.secret_managers.main import get_secret_str from litellm.types.rerank import ( + HostedVLLMRerankTruncationParams, OptionalRerankParams, RerankBilledUnits, RerankRequest, @@ -62,7 +64,11 @@ class HostedVLLMRerankConfig(BaseRerankConfig): "top_n", "rank_fields", "return_documents", + "max_tokens_per_doc", "instruction", + "truncate_prompt_tokens", + "truncation_side", + "max_tokens_per_query", ] def map_cohere_rerank_params( @@ -100,7 +106,15 @@ class HostedVLLMRerankConfig(BaseRerankConfig): if instruction is not None: mapped_params["instruction"] = instruction - return dict(mapped_params) + truncation: Final = HostedVLLMRerankTruncationParams.model_validate(non_default_params or MappingProxyType({})) + forwarded: Final[OptionalRerankParams] = { + **mapped_params, + "max_tokens_per_doc": max_tokens_per_doc, + "truncate_prompt_tokens": truncation.truncate_prompt_tokens, + "truncation_side": truncation.truncation_side, + "max_tokens_per_query": truncation.max_tokens_per_query, + } + return dict(forwarded) def validate_environment( self, @@ -138,6 +152,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig): if "documents" not in optional_rerank_params: raise ValueError("documents is required for Hosted VLLM rerank") + truncation: Final = HostedVLLMRerankTruncationParams.model_validate(optional_rerank_params) rerank_request: Final = RerankRequest( model=model, query=optional_rerank_params["query"], @@ -146,6 +161,10 @@ class HostedVLLMRerankConfig(BaseRerankConfig): rank_fields=optional_rerank_params.get("rank_fields", None), return_documents=optional_rerank_params.get("return_documents", None), instruction=optional_rerank_params.get("instruction", None), + max_tokens_per_doc=truncation.max_tokens_per_doc, + truncate_prompt_tokens=truncation.truncate_prompt_tokens, + truncation_side=truncation.truncation_side, + max_tokens_per_query=truncation.max_tokens_per_query, ) return rerank_request.model_dump(exclude_none=True) diff --git a/litellm/types/rerank.py b/litellm/types/rerank.py index 903781b2ccd..a76e6cf1187 100644 --- a/litellm/types/rerank.py +++ b/litellm/types/rerank.py @@ -4,8 +4,10 @@ https://docs.cohere.com/reference/rerank """ -from pydantic import BaseModel, PrivateAttr -from typing_extensions import Required, TypedDict +from typing import Literal + +from pydantic import BaseModel, ConfigDict, PrivateAttr +from typing_extensions import ReadOnly, Required, TypedDict class RerankRequest(BaseModel): @@ -21,6 +23,18 @@ class RerankRequest(BaseModel): # (e.g. hosted vLLM / Qwen3-Reranker, DeepInfra). Omitted from the outgoing # request when None, so this is fully backward-compatible. instruction: str | None = None + truncate_prompt_tokens: int | None = None + truncation_side: Literal["left", "right"] | None = None + max_tokens_per_query: int | None = None + + +class HostedVLLMRerankTruncationParams(BaseModel): + model_config = ConfigDict(frozen=True) + + truncate_prompt_tokens: int | None = None + truncation_side: Literal["left", "right"] | None = None + max_tokens_per_query: int | None = None + max_tokens_per_doc: int | None = None class OptionalRerankParams(TypedDict, total=False): @@ -32,6 +46,9 @@ class OptionalRerankParams(TypedDict, total=False): max_chunks_per_doc: int | None max_tokens_per_doc: int | None instruction: str | None + truncate_prompt_tokens: ReadOnly[int | None] + truncation_side: ReadOnly[Literal["left", "right"] | None] + max_tokens_per_query: ReadOnly[int | None] class RerankBilledUnits(TypedDict, total=False): diff --git a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py index e6e6aa946d5..da27ea1ac58 100644 --- a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py @@ -1,8 +1,13 @@ +import json import os import sys +from unittest.mock import MagicMock, patch +import httpx import pytest +import litellm +from litellm.llms.custom_httpx.http_handler import HTTPHandler from litellm.llms.hosted_vllm.rerank.transformation import HostedVLLMRerankConfig from litellm.rerank_api.rerank_utils import get_optional_rerank_params from litellm.types.rerank import ( @@ -87,9 +92,7 @@ class TestHostedVLLMRerankTransform: assert "instruction" not in body def test_map_cohere_rerank_params_raises_on_max_chunks_per_doc(self): - with pytest.raises( - ValueError, match="Hosted VLLM does not support max_chunks_per_doc" - ): + with pytest.raises(ValueError, match="Hosted VLLM does not support max_chunks_per_doc"): self.config.map_cohere_rerank_params( non_default_params=None, model=self.model, @@ -104,12 +107,10 @@ class TestHostedVLLMRerankTransform: url = self.config.get_complete_url(base, self.model) assert url == "https://api.example.com/rerank" # Already ends with /rerank - url2 = self.config.get_complete_url( - "https://api.example.com/rerank", self.model - ) + url2 = self.config.get_complete_url("https://api.example.com/rerank", self.model) assert url2 == "https://api.example.com/rerank" # Raises if api_base is None - with pytest.raises(ValueError, match='api_base must be provided for Hosted VLLM rerank'): + with pytest.raises(ValueError, match="api_base must be provided for Hosted VLLM rerank"): self.config.get_complete_url(None, self.model) def test_transform_response(self): @@ -173,3 +174,110 @@ class TestGetOptionalRerankParamsInstruction: documents=["doc1", "doc2"], ) assert "instruction" not in params + + +class TestHostedVLLMRerankTruncationParams: + def setup_method(self): + self.config = HostedVLLMRerankConfig() + self.model = "hosted-vllm-model" + + def test_map_cohere_rerank_params_forwards_vllm_truncation_params(self): + params = self.config.map_cohere_rerank_params( + non_default_params={ + "truncate_prompt_tokens": 512, + "truncation_side": "left", + "max_tokens_per_query": 64, + "metadata": {"user_api_key": "sk-test"}, + }, + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + max_tokens_per_doc=128, + ) + assert params["truncate_prompt_tokens"] == 512 + assert params["truncation_side"] == "left" + assert params["max_tokens_per_query"] == 64 + assert params["max_tokens_per_doc"] == 128 + assert "metadata" not in params + + def test_map_cohere_rerank_params_omits_truncation_params_when_absent(self): + params = self.config.map_cohere_rerank_params( + non_default_params={"metadata": {"user_api_key": "sk-test"}}, + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + ) + body = self.config.transform_rerank_request(model=self.model, optional_rerank_params=params, headers={}) + truncation_keys = {"truncate_prompt_tokens", "truncation_side", "max_tokens_per_query", "max_tokens_per_doc"} + assert not truncation_keys & body.keys() + assert body == { + "model": self.model, + "query": "test query", + "documents": ["doc1", "doc2"], + "return_documents": True, + } + + def test_map_cohere_rerank_params_rejects_invalid_truncation_side(self): + with pytest.raises(ValueError, match="truncation_side"): + self.config.map_cohere_rerank_params( + non_default_params={"truncation_side": "middle"}, + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + ) + + def test_transform_request_forwards_truncation_params(self): + body = self.config.transform_rerank_request( + model=self.model, + optional_rerank_params={ + "query": "test query", + "documents": ["doc1", "doc2"], + "truncate_prompt_tokens": 512, + "truncation_side": "left", + "max_tokens_per_query": 64, + "max_tokens_per_doc": 128, + }, + headers={}, + ) + assert body["truncate_prompt_tokens"] == 512 + assert body["truncation_side"] == "left" + assert body["max_tokens_per_query"] == 64 + assert body["max_tokens_per_doc"] == 128 + + def test_transform_request_omits_truncation_params_when_absent(self): + body = self.config.transform_rerank_request( + model=self.model, + optional_rerank_params={"query": "test query", "documents": ["doc1", "doc2"]}, + headers={}, + ) + assert "truncate_prompt_tokens" not in body + assert "truncation_side" not in body + assert "max_tokens_per_query" not in body + assert "max_tokens_per_doc" not in body + + def test_rerank_sends_truncate_prompt_tokens_to_vllm(self): + client = HTTPHandler() + mock_response = MagicMock(spec=httpx.Response) + mock_response.status_code = 200 + mock_response.json.return_value = { + "id": "score-1", + "results": [{"index": 0, "relevance_score": 0.5}], + "usage": {"total_tokens": 512}, + } + with patch.object(client, "post", return_value=mock_response) as mock_post: + litellm.rerank( + model="hosted_vllm/BAAI/bge-reranker-base", + api_base="http://vllm.local:8000", + query="List all the unique case ids", + documents=["a document longer than the reranker context window"], + truncate_prompt_tokens=512, + truncation_side="left", + client=client, + ) + sent_body = json.loads(mock_post.call_args.kwargs["data"]) + assert mock_post.call_args.kwargs["url"] == "http://vllm.local:8000/rerank" + assert sent_body["truncate_prompt_tokens"] == 512 + assert sent_body["truncation_side"] == "left" From dfaf23523453de12278e3d30801075c4da6ee903 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:08:45 -0700 Subject: [PATCH 19/31] fix(bedrock): honor BEDROCK_MANTLE_API_BASE on bedrock/mantle messages and chat URLs --- litellm/llms/bedrock/common_utils.py | 7 ++-- .../test_litellm/llms/bedrock/test_mantle.py | 42 +++++++++++++++++++ 2 files changed, 46 insertions(+), 3 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 66ee5f10679..048d023a1bd 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -758,12 +758,13 @@ def build_mantle_messages_url( """Build the bedrock-mantle Anthropic /messages URL. Honors an explicit endpoint override (``api_base``, then - ``aws_bedrock_runtime_endpoint``) so private VPC / VPCE / GovCloud Mantle - endpoints are reachable; otherwise falls back to the public regional host. + ``aws_bedrock_runtime_endpoint``, then ``BEDROCK_MANTLE_API_BASE``) so + private VPC / VPCE / GovCloud Mantle endpoints are reachable; otherwise + falls back to the public regional host. The mantle messages path is appended unless the override already carries it, so callers can pass either the host or the full messages URL. """ - override: Final = api_base or aws_bedrock_runtime_endpoint + override: Final = api_base or aws_bedrock_runtime_endpoint or get_secret_str("BEDROCK_MANTLE_API_BASE") if override: base: Final = override.rstrip("/") if base.endswith(MANTLE_MESSAGES_PATH): diff --git a/tests/test_litellm/llms/bedrock/test_mantle.py b/tests/test_litellm/llms/bedrock/test_mantle.py index d34517f61f6..d1d1ba447fb 100644 --- a/tests/test_litellm/llms/bedrock/test_mantle.py +++ b/tests/test_litellm/llms/bedrock/test_mantle.py @@ -128,6 +128,12 @@ def test_mantle_messages_url_construction(): _VPC_ENDPOINT = "https://vpce-0a1b2c3d.bedrock-mantle.us-gov-west-1.vpce.amazonaws.com" +@pytest.fixture(autouse=True) +def no_ambient_mantle_api_base(monkeypatch): + monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False) + + + def test_mantle_chat_url_honors_api_base_host(): config = AmazonMantleConfig() url = config.get_complete_url( @@ -193,6 +199,42 @@ def test_mantle_messages_url_honors_aws_bedrock_runtime_endpoint(): assert url == f"{_VPC_ENDPOINT}/anthropic/v1/messages" +_ENV_ENDPOINT = "https://bedrock-mantle.us-east-1.api.aws.internal.example.com" + + +@pytest.mark.parametrize("config_cls", [AmazonMantleConfig, AmazonMantleMessagesConfig]) +def test_mantle_url_honors_bedrock_mantle_api_base_env(monkeypatch, config_cls): + monkeypatch.setenv("BEDROCK_MANTLE_API_BASE", _ENV_ENDPOINT) + url = config_cls().get_complete_url( + api_base=None, + api_key=None, + model="mantle/anthropic.claude-mythos-preview", + optional_params={"aws_region_name": "us-east-1"}, + litellm_params={}, + ) + assert url == f"{_ENV_ENDPOINT}/anthropic/v1/messages" + + +@pytest.mark.parametrize("config_cls", [AmazonMantleConfig, AmazonMantleMessagesConfig]) +@pytest.mark.parametrize( + ("api_base", "optional_params"), + [ + (_VPC_ENDPOINT, {"aws_region_name": "us-gov-west-1"}), + (None, {"aws_region_name": "us-gov-west-1", "aws_bedrock_runtime_endpoint": _VPC_ENDPOINT}), + ], +) +def test_mantle_url_explicit_endpoint_beats_bedrock_mantle_api_base_env(monkeypatch, config_cls, api_base, optional_params): + monkeypatch.setenv("BEDROCK_MANTLE_API_BASE", _ENV_ENDPOINT) + url = config_cls().get_complete_url( + api_base=api_base, + api_key=None, + model="mantle/anthropic.claude-mythos-preview", + optional_params=optional_params, + litellm_params={}, + ) + assert url == f"{_VPC_ENDPOINT}/anthropic/v1/messages" + + def test_mantle_transform_request_strips_prefix_and_adds_model(): config = AmazonMantleConfig() request = config.transform_request( From 8d00220acef335234797b00cf3b1f1ee73db2b3a Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:19:20 -0700 Subject: [PATCH 20/31] test(hosted_vllm): annotate rerank truncation test locals as Final --- .../test_hosted_vllm_rerank_transformation.py | 24 ++++++++++++------- 1 file changed, 15 insertions(+), 9 deletions(-) diff --git a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py index da27ea1ac58..49d58cc28c4 100644 --- a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py @@ -1,6 +1,7 @@ import json import os import sys +from typing import Final from unittest.mock import MagicMock, patch import httpx @@ -182,7 +183,7 @@ class TestHostedVLLMRerankTruncationParams: self.model = "hosted-vllm-model" def test_map_cohere_rerank_params_forwards_vllm_truncation_params(self): - params = self.config.map_cohere_rerank_params( + params: Final = self.config.map_cohere_rerank_params( non_default_params={ "truncate_prompt_tokens": 512, "truncation_side": "left", @@ -202,15 +203,20 @@ class TestHostedVLLMRerankTruncationParams: assert "metadata" not in params def test_map_cohere_rerank_params_omits_truncation_params_when_absent(self): - params = self.config.map_cohere_rerank_params( + params: Final = self.config.map_cohere_rerank_params( non_default_params={"metadata": {"user_api_key": "sk-test"}}, model=self.model, drop_params=False, query="test query", documents=["doc1", "doc2"], ) - body = self.config.transform_rerank_request(model=self.model, optional_rerank_params=params, headers={}) - truncation_keys = {"truncate_prompt_tokens", "truncation_side", "max_tokens_per_query", "max_tokens_per_doc"} + body: Final = self.config.transform_rerank_request(model=self.model, optional_rerank_params=params, headers={}) + truncation_keys: Final = { + "truncate_prompt_tokens", + "truncation_side", + "max_tokens_per_query", + "max_tokens_per_doc", + } assert not truncation_keys & body.keys() assert body == { "model": self.model, @@ -230,7 +236,7 @@ class TestHostedVLLMRerankTruncationParams: ) def test_transform_request_forwards_truncation_params(self): - body = self.config.transform_rerank_request( + body: Final = self.config.transform_rerank_request( model=self.model, optional_rerank_params={ "query": "test query", @@ -248,7 +254,7 @@ class TestHostedVLLMRerankTruncationParams: assert body["max_tokens_per_doc"] == 128 def test_transform_request_omits_truncation_params_when_absent(self): - body = self.config.transform_rerank_request( + body: Final = self.config.transform_rerank_request( model=self.model, optional_rerank_params={"query": "test query", "documents": ["doc1", "doc2"]}, headers={}, @@ -259,8 +265,8 @@ class TestHostedVLLMRerankTruncationParams: assert "max_tokens_per_doc" not in body def test_rerank_sends_truncate_prompt_tokens_to_vllm(self): - client = HTTPHandler() - mock_response = MagicMock(spec=httpx.Response) + client: Final = HTTPHandler() + mock_response: Final = MagicMock(spec=httpx.Response) mock_response.status_code = 200 mock_response.json.return_value = { "id": "score-1", @@ -277,7 +283,7 @@ class TestHostedVLLMRerankTruncationParams: truncation_side="left", client=client, ) - sent_body = json.loads(mock_post.call_args.kwargs["data"]) + sent_body: Final = json.loads(mock_post.call_args.kwargs["data"]) assert mock_post.call_args.kwargs["url"] == "http://vllm.local:8000/rerank" assert sent_body["truncate_prompt_tokens"] == 512 assert sent_body["truncation_side"] == "left" From ef14bed0296b4a6c48777b8f9df8018b11179c4b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:31:19 -0700 Subject: [PATCH 21/31] fix(hosted_vllm): reject invalid rerank truncation params with a 400 --- .../llms/hosted_vllm/rerank/transformation.py | 11 +++++++- .../test_hosted_vllm_rerank_transformation.py | 26 ++++++++++++------- 2 files changed, 26 insertions(+), 11 deletions(-) diff --git a/litellm/llms/hosted_vllm/rerank/transformation.py b/litellm/llms/hosted_vllm/rerank/transformation.py index 265eb350fc6..764d80c6f82 100644 --- a/litellm/llms/hosted_vllm/rerank/transformation.py +++ b/litellm/llms/hosted_vllm/rerank/transformation.py @@ -7,8 +7,10 @@ from types import MappingProxyType from typing import Any, Final import httpx +from pydantic import ValidationError from litellm._uuid import uuid +from litellm.exceptions import UnsupportedParamsError from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.base_llm.rerank.transformation import BaseRerankConfig @@ -36,6 +38,13 @@ class HostedVLLMRerankError(BaseLLMException): super().__init__(status_code=status_code, message=message, headers=headers) +def validated_truncation_params(non_default_params: Mapping[str, object] | None) -> HostedVLLMRerankTruncationParams: + try: + return HostedVLLMRerankTruncationParams.model_validate(non_default_params or MappingProxyType({})) + except ValidationError as error: + raise UnsupportedParamsError(status_code=400, message=f"hosted_vllm rerank: {error}") from error + + class HostedVLLMRerankConfig(BaseRerankConfig): def __init__(self) -> None: pass @@ -106,7 +115,7 @@ class HostedVLLMRerankConfig(BaseRerankConfig): if instruction is not None: mapped_params["instruction"] = instruction - truncation: Final = HostedVLLMRerankTruncationParams.model_validate(non_default_params or MappingProxyType({})) + truncation: Final = validated_truncation_params(non_default_params) forwarded: Final[OptionalRerankParams] = { **mapped_params, "max_tokens_per_doc": max_tokens_per_doc, diff --git a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py index 49d58cc28c4..9a62fcf6f0f 100644 --- a/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/test_hosted_vllm_rerank_transformation.py @@ -202,6 +202,22 @@ class TestHostedVLLMRerankTruncationParams: assert params["max_tokens_per_doc"] == 128 assert "metadata" not in params + @pytest.mark.parametrize( + "bad_params", + [{"truncation_side": "middle"}, {"truncate_prompt_tokens": "lots"}, {"max_tokens_per_query": -1.5}], + ) + def test_map_cohere_rerank_params_rejects_invalid_truncation_params_as_400(self, bad_params: dict[str, object]): + with pytest.raises(litellm.UnsupportedParamsError) as raised: + self.config.map_cohere_rerank_params( + non_default_params=dict(bad_params), + model=self.model, + drop_params=False, + query="test query", + documents=["doc1", "doc2"], + ) + assert raised.value.status_code == 400 + assert next(iter(bad_params)) in str(raised.value) + def test_map_cohere_rerank_params_omits_truncation_params_when_absent(self): params: Final = self.config.map_cohere_rerank_params( non_default_params={"metadata": {"user_api_key": "sk-test"}}, @@ -225,16 +241,6 @@ class TestHostedVLLMRerankTruncationParams: "return_documents": True, } - def test_map_cohere_rerank_params_rejects_invalid_truncation_side(self): - with pytest.raises(ValueError, match="truncation_side"): - self.config.map_cohere_rerank_params( - non_default_params={"truncation_side": "middle"}, - model=self.model, - drop_params=False, - query="test query", - documents=["doc1", "doc2"], - ) - def test_transform_request_forwards_truncation_params(self): body: Final = self.config.transform_rerank_request( model=self.model, From 49c69c46b25af2dd962b322bd6c8e6c5668546c6 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:51:27 -0700 Subject: [PATCH 22/31] fix(bedrock): drop the OpenAI base suffix from BEDROCK_MANTLE_API_BASE before the mantle messages path --- litellm/llms/bedrock/common_utils.py | 15 +++++++++++++-- tests/test_litellm/llms/bedrock/test_mantle.py | 12 +++++++++--- 2 files changed, 22 insertions(+), 5 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 048d023a1bd..1e5329c90dd 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -748,6 +748,15 @@ def strip_bedrock_throughput_suffix(model: str) -> str: MANTLE_MESSAGES_PATH: Final = "/anthropic/v1/messages" +_MANTLE_OPENAI_BASE_SUFFIXES: Final = ("/openai/v1", "/v1") + + +def _mantle_api_base_from_env() -> str | None: + env_base: Final = get_secret_str("BEDROCK_MANTLE_API_BASE") + if env_base is None: + return None + base: Final = env_base.rstrip("/") + return next((base[: -len(suffix)] for suffix in _MANTLE_OPENAI_BASE_SUFFIXES if base.endswith(suffix)), base) def build_mantle_messages_url( @@ -762,9 +771,11 @@ def build_mantle_messages_url( private VPC / VPCE / GovCloud Mantle endpoints are reachable; otherwise falls back to the public regional host. The mantle messages path is appended unless the override already carries it, - so callers can pass either the host or the full messages URL. + so callers can pass either the host or the full messages URL. The env var is + shared with the OpenAI-surface ``bedrock_mantle/*`` routes, which need it to + carry their ``/v1`` or ``/openai/v1`` base, so that suffix is dropped first. """ - override: Final = api_base or aws_bedrock_runtime_endpoint or get_secret_str("BEDROCK_MANTLE_API_BASE") + override: Final = api_base or aws_bedrock_runtime_endpoint or _mantle_api_base_from_env() if override: base: Final = override.rstrip("/") if base.endswith(MANTLE_MESSAGES_PATH): diff --git a/tests/test_litellm/llms/bedrock/test_mantle.py b/tests/test_litellm/llms/bedrock/test_mantle.py index d1d1ba447fb..09be2118001 100644 --- a/tests/test_litellm/llms/bedrock/test_mantle.py +++ b/tests/test_litellm/llms/bedrock/test_mantle.py @@ -203,8 +203,12 @@ _ENV_ENDPOINT = "https://bedrock-mantle.us-east-1.api.aws.internal.example.com" @pytest.mark.parametrize("config_cls", [AmazonMantleConfig, AmazonMantleMessagesConfig]) -def test_mantle_url_honors_bedrock_mantle_api_base_env(monkeypatch, config_cls): - monkeypatch.setenv("BEDROCK_MANTLE_API_BASE", _ENV_ENDPOINT) +@pytest.mark.parametrize( + "env_value", + [_ENV_ENDPOINT, f"{_ENV_ENDPOINT}/", f"{_ENV_ENDPOINT}/v1", f"{_ENV_ENDPOINT}/openai/v1"], +) +def test_mantle_url_honors_bedrock_mantle_api_base_env(monkeypatch, config_cls, env_value): + monkeypatch.setenv("BEDROCK_MANTLE_API_BASE", env_value) url = config_cls().get_complete_url( api_base=None, api_key=None, @@ -223,7 +227,9 @@ def test_mantle_url_honors_bedrock_mantle_api_base_env(monkeypatch, config_cls): (None, {"aws_region_name": "us-gov-west-1", "aws_bedrock_runtime_endpoint": _VPC_ENDPOINT}), ], ) -def test_mantle_url_explicit_endpoint_beats_bedrock_mantle_api_base_env(monkeypatch, config_cls, api_base, optional_params): +def test_mantle_url_explicit_endpoint_beats_bedrock_mantle_api_base_env( + monkeypatch, config_cls, api_base, optional_params +): monkeypatch.setenv("BEDROCK_MANTLE_API_BASE", _ENV_ENDPOINT) url = config_cls().get_complete_url( api_base=api_base, From 9bd870d47a700b183e0ee0d9bf7647aa7c739561 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:36:28 -0700 Subject: [PATCH 23/31] fix(databricks): upgrade legacy thinking to adaptive on adaptive-only Claude models --- .../llms/databricks/chat/transformation.py | 4 ++++ .../test_databricks_chat_transformation.py | 21 +++++++++++++++++++ 2 files changed, 25 insertions(+) diff --git a/litellm/llms/databricks/chat/transformation.py b/litellm/llms/databricks/chat/transformation.py index c587146005f..65622d62af2 100644 --- a/litellm/llms/databricks/chat/transformation.py +++ b/litellm/llms/databricks/chat/transformation.py @@ -330,6 +330,10 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig): ) -> dict: is_thinking_enabled: Final = self.is_thinking_enabled(non_default_params) mapped_params: Final = super().map_openai_params(non_default_params, optional_params, model, drop_params) + if "claude" in model: + AnthropicConfig.translate_legacy_thinking_for_adaptive_model( + model=model, optional_params=mapped_params, custom_llm_provider="databricks" + ) if "tools" in mapped_params: mapped_params["tools"] = self._map_openai_to_dbrx_tool(model=model, tools=mapped_params["tools"]) if "max_completion_tokens" in non_default_params and replace_max_completion_tokens_with_max_tokens: diff --git a/tests/test_litellm/llms/databricks/chat/test_databricks_chat_transformation.py b/tests/test_litellm/llms/databricks/chat/test_databricks_chat_transformation.py index 41fb2589655..71661cc532b 100644 --- a/tests/test_litellm/llms/databricks/chat/test_databricks_chat_transformation.py +++ b/tests/test_litellm/llms/databricks/chat/test_databricks_chat_transformation.py @@ -422,6 +422,27 @@ def test_databricks_config_probes_capabilities_under_databricks_namespace(): assert DatabricksConfig().custom_llm_provider == "databricks" +@pytest.mark.parametrize( + "model, expected_thinking, expected_output_config", + [ + ("databricks-claude-opus-4-8", {"type": "adaptive"}, {"effort": "high"}), + ("databricks-claude-opus-4-6", {"type": "enabled", "budget_tokens": 4096}, None), + ], + ids=["adaptive_only_upgrades_to_adaptive", "legacy_capable_forwards_verbatim"], +) +def test_map_openai_params_upgrades_legacy_thinking_on_adaptive_only_claude( + model, expected_thinking, expected_output_config +): + mapped = DatabricksConfig().map_openai_params( + non_default_params={"thinking": {"type": "enabled", "budget_tokens": 4096}}, + optional_params={}, + model=model, + drop_params=False, + ) + assert mapped["thinking"] == expected_thinking + assert mapped.get("output_config") == expected_output_config + + def _streaming_chunk(usage=None, choices=None): base = { "id": "chatcmpl-test", From 0346bb265934a09bd5d8eab336facba1cc5bc01b Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:58:05 +0000 Subject: [PATCH 24/31] fix(bedrock): upgrade legacy thinking after the invoke response_format stub model swap Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../anthropic_claude3_transformation.py | 5 +++++ ...ations_anthropic_claude3_transformation.py | 19 +++++++++++++++++++ 2 files changed, 24 insertions(+) diff --git a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py index 2a4c38e71ea..07ddf6570f2 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py @@ -107,6 +107,11 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): # Restore original model name model = original_model + # The stub model hides the original model from the parent's legacy thinking upgrade + AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model( + model=original_model, optional_params=optional_params, custom_llm_provider="bedrock" + ) + # The stub model hides the original model from the parent's forced-tool-use backstop response_format_tool_choice: Final = optional_params.get("tool_choice") if ( diff --git a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py index 41d82e4f960..d136cb6450b 100644 --- a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py @@ -671,3 +671,22 @@ def test_bedrock_chat_invoke_fable_5_1_response_format_avoids_forced_tool_choice assert "output_format" not in result assert "tools" in result assert "tool_choice" not in result + + +def test_bedrock_chat_invoke_response_format_stub_still_upgrades_legacy_thinking(local_model_cost_map): + """Regression: the tool-based ``response_format`` path swaps in a Claude 3 stub + model before the shared Anthropic mapping, which hid the adaptive-only model + from the legacy ``thinking`` upgrade and left ``type=enabled`` on the wire.""" + result = AmazonAnthropicClaudeConfig().map_openai_params( + non_default_params={ + "response_format": {"type": "json_object"}, + "thinking": {"type": "enabled", "budget_tokens": 4096}, + "max_tokens": 8192, + }, + optional_params={}, + model="us.anthropic.claude-fable-5-1", + drop_params=False, + ) + + assert result["thinking"] == {"type": "adaptive"} + assert result["output_config"] == {"effort": "high"} From fd4b15fae6d8592aacd1ee2a60cdb9738f5dfc4e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:37:14 -0700 Subject: [PATCH 25/31] fix(anthropic): upgrade legacy thinking after the Bedrock Invoke and Vertex structured-output stub swap --- .../anthropic_claude3_transformation.py | 4 ++++ .../anthropic/transformation.py | 4 ++++ ...ations_anthropic_claude3_transformation.py | 24 +++++++++++++++++++ ...partner_models_anthropic_transformation.py | 23 ++++++++++++++++++ 4 files changed, 55 insertions(+) diff --git a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py index 2a4c38e71ea..67720451c00 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py @@ -107,6 +107,10 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): # Restore original model name model = original_model + AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model( + model=original_model, optional_params=optional_params, custom_llm_provider="bedrock" + ) + # The stub model hides the original model from the parent's forced-tool-use backstop response_format_tool_choice: Final = optional_params.get("tool_choice") if ( diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index ef03e61a858..7579bc8c02e 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -177,6 +177,10 @@ class VertexAIAnthropicConfig(AnthropicConfig): # Restore original model name for any other processing model = original_model + AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model( + model=original_model, optional_params=optional_params, custom_llm_provider="vertex_ai" + ) + return optional_params def transform_response( diff --git a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py index 41d82e4f960..4a1e97ca170 100644 --- a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py @@ -671,3 +671,27 @@ def test_bedrock_chat_invoke_fable_5_1_response_format_avoids_forced_tool_choice assert "output_format" not in result assert "tools" in result assert "tool_choice" not in result + + +@pytest.mark.parametrize("model", ["us.anthropic.claude-sonnet-5", "us.anthropic.claude-fable-5-1"]) +def test_bedrock_chat_invoke_tool_based_response_format_still_upgrades_legacy_thinking(local_model_cost_map, model): + result = AmazonAnthropicClaudeConfig().map_openai_params( + non_default_params={ + "response_format": { + "type": "json_schema", + "json_schema": { + "name": "test_schema", + "schema": {"type": "object", "properties": {"result": {"type": "string"}}}, + }, + }, + "thinking": {"type": "enabled", "budget_tokens": 4096}, + "max_tokens": 8192, + }, + optional_params={}, + model=model, + drop_params=False, + ) + + assert "tools" in result + assert result["thinking"] == {"type": "adaptive"} + assert result["output_config"] == {"effort": "high"} diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 9419f88a981..a57672cfbfb 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -752,3 +752,26 @@ def test_vertex_ai_fable_5_1_response_format_uses_native_output_format(local_mod assert "output_format" in result_params assert "tool_choice" not in result_params assert "tools" not in result_params + + +def test_vertex_ai_anthropic_tool_based_response_format_still_upgrades_legacy_thinking(local_model_cost_map): + result_params = VertexAIAnthropicConfig().map_openai_params( + non_default_params={ + "response_format": { + "type": "json_schema", + "json_schema": { + "name": "test_schema", + "schema": {"type": "object", "properties": {"result": {"type": "string"}}}, + }, + }, + "thinking": {"type": "enabled", "budget_tokens": 4096}, + "max_tokens": 8192, + }, + optional_params={}, + model="claude-opus-4-8", + drop_params=False, + ) + + assert "tools" in result_params + assert result_params["thinking"] == {"type": "adaptive"} + assert result_params["output_config"] == {"effort": "high"} From e3f61ce1347286774bc7c5da301c2bda1d0b0a96 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:13:23 +0000 Subject: [PATCH 26/31] chore(techdebt): refresh lint budgets after staging merge Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- basedpyright-code-budget.json | 10 +++++----- type-discipline-budget.json | 10 +++++----- 2 files changed, 10 insertions(+), 10 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index 64a22209d86..a53d7976263 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5601 }, "reportMissingTypeArgument": { - "limit": 15298 + "limit": 15296 }, "reportMissingTypeStubs": { "limit": 40 @@ -105,14 +105,14 @@ "limit": 109 }, "reportUnknownMemberType": { - "limit": 38344 + "limit": 38341 }, "reportUnknownParameterType": { "limit": 19625 }, - "reportUnknownVariableType": { - "limit": 29877 - }, + "reportUnknownVariableType": { + "limit": 29873 + }, "reportUnnecessaryCast": { "limit": 111 }, diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 5823d83d0bd..5475fc848fe 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -6,7 +6,7 @@ "limit": 26777 }, "LIT003": { - "limit": 265 + "limit": 264 }, "LIT004": { "limit": 40 @@ -26,11 +26,11 @@ "LIT009": { "limit": 0 }, - "LIT010": { - "limit": 16494 - }, + "LIT010": { + "limit": 16492 + }, "LIT011": { - "limit": 5529 + "limit": 5527 }, "LIT012": { "limit": 4495 From 6265437592e9d2401cae1149b9747e2eb2082572 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:35:02 +0000 Subject: [PATCH 27/31] chore(techdebt): avoid new mutable literal in gigachat tool conversion Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- basedpyright-code-budget.json | 6 +++--- litellm/llms/gigachat/chat/transformation.py | 8 +++++++- type-discipline-budget.json | 6 +++--- 3 files changed, 13 insertions(+), 7 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index 449ed5bfa1e..e03ed8abdca 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5601 }, "reportMissingTypeArgument": { - "limit": 15296 + "limit": 15292 }, "reportMissingTypeStubs": { "limit": 40 @@ -105,13 +105,13 @@ "limit": 109 }, "reportUnknownMemberType": { - "limit": 38341 + "limit": 38335 }, "reportUnknownParameterType": { "limit": 19625 }, "reportUnknownVariableType": { - "limit": 29873 + "limit": 29865 }, "reportUnnecessaryCast": { "limit": 111 diff --git a/litellm/llms/gigachat/chat/transformation.py b/litellm/llms/gigachat/chat/transformation.py index 991a93ccb21..89920ebd27b 100644 --- a/litellm/llms/gigachat/chat/transformation.py +++ b/litellm/llms/gigachat/chat/transformation.py @@ -10,6 +10,7 @@ import json import time import uuid from collections.abc import AsyncIterator, Iterator, Mapping, Sequence +from types import MappingProxyType from typing import TYPE_CHECKING, Any, Final import httpx @@ -34,6 +35,9 @@ else: LiteLLMLoggingObj = Any +_EMPTY_FUNCTION: Final[Mapping[str, object]] = MappingProxyType({}) + + def is_valid_json(value: str) -> bool: """Checks whether the value passed is a valid serialized JSON string""" try: @@ -213,7 +217,9 @@ class GigaChatConfig(BaseConfig): "parameters": function.get("parameters", {}), } for function in ( - tool.get("function", {}) for tool in tools if isinstance(tool, dict) and tool.get("type") == "function" + tool.get("function", _EMPTY_FUNCTION) + for tool in tools + if isinstance(tool, dict) and tool.get("type") == "function" ) ] diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 5a4cae71d41..45d00a71d39 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -6,7 +6,7 @@ "limit": 26774 }, "LIT003": { - "limit": 264 + "limit": 262 }, "LIT004": { "limit": 40 @@ -27,10 +27,10 @@ "limit": 0 }, "LIT010": { - "limit": 16492 + "limit": 16488 }, "LIT011": { - "limit": 5527 + "limit": 5523 }, "LIT012": { "limit": 4495 From 683fc340440f4c6003075ec7fe42451e0875e734 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:23:52 -0700 Subject: [PATCH 28/31] fix(azure_ai): let the caller's output_config from extra_body win over the legacy thinking upgrade --- .../llms/azure_ai/anthropic/transformation.py | 7 +++-- .../test_azure_anthropic_transformation.py | 30 +++++++++++++++++-- 2 files changed, 31 insertions(+), 6 deletions(-) diff --git a/litellm/llms/azure_ai/anthropic/transformation.py b/litellm/llms/azure_ai/anthropic/transformation.py index c5053448627..864d2134a84 100644 --- a/litellm/llms/azure_ai/anthropic/transformation.py +++ b/litellm/llms/azure_ai/anthropic/transformation.py @@ -17,13 +17,14 @@ def _promote_extra_body_to_optional_params(optional_params: dict) -> None: ``output_config`` get auto-routed into ``extra_body`` by ``add_provider_specific_params_to_optional_params``. For the Azure→Anthropic route those keys must reach the request body and be validated, so promote - them. ``setdefault`` keeps explicit top-level values authoritative. + them. The caller's values overwrite mapped top-level duplicates, matching + the native ``anthropic`` provider, where the same passthrough lands on + top-level ``optional_params`` after mapping. """ extra_body: Final = optional_params.get("extra_body") if not isinstance(extra_body, dict) or not extra_body: return - for k, v in extra_body.items(): - optional_params.setdefault(k, v) + optional_params.update(extra_body) optional_params.pop("extra_body", None) diff --git a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py index 9dac914ca4d..9e2bfb08852 100644 --- a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py +++ b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py @@ -362,8 +362,8 @@ class TestAzureAnthropicConfig: ) assert "xhigh" in str(exc_info.value) - def test_extra_body_promotion_does_not_clobber_top_level(self): - """Top-level ``optional_params`` wins over duplicates in ``extra_body``.""" + def test_extra_body_promotion_overrides_mapped_top_level(self): + """The caller's ``extra_body`` wins over a mapped top-level duplicate, like the native ``anthropic`` passthrough.""" config = AzureAnthropicConfig() messages = [{"role": "user", "content": "Hello"}] @@ -383,7 +383,31 @@ class TestAzureAnthropicConfig: headers=headers, ) - assert result["output_config"] == {"effort": "low"} + assert result["output_config"] == {"effort": "high"} + + def test_legacy_thinking_upgrade_keeps_caller_effort_from_extra_body(self, local_model_cost_map): + config = AzureAnthropicConfig() + + mapped = config.map_openai_params( + non_default_params={"thinking": {"type": "enabled", "budget_tokens": 1024}, "max_tokens": 100}, + optional_params={}, + model="claude-opus-4-8", + drop_params=False, + ) + assert mapped["thinking"] == {"type": "adaptive"} + assert mapped["output_config"] == {"effort": "low"} + + result = config.transform_request( + model="claude-opus-4-8", + messages=[{"role": "user", "content": "Hello"}], + optional_params={**mapped, "extra_body": {"output_config": {"effort": "high"}}}, + litellm_params={"api_key": "test-key"}, + headers={"api-key": "test-key", "anthropic-version": "2023-06-01"}, + ) + + assert result["thinking"] == {"type": "adaptive"} + assert result["output_config"] == {"effort": "high"} + assert "extra_body" not in result def test_context_management_mixed_edits_beta_headers(self): """Test that context_management with both compact and other edits adds both beta headers""" From 63f5de12ae4e21f4cc49bf2ac23a89dc077b1854 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:25:58 -0700 Subject: [PATCH 29/31] chore(techdebt): ratchet lint budgets to the merged tree --- basedpyright-code-budget.json | 6 +++--- type-discipline-budget.json | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index e03ed8abdca..7d3c83a567f 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -57,7 +57,7 @@ "limit": 5601 }, "reportMissingTypeArgument": { - "limit": 15292 + "limit": 15290 }, "reportMissingTypeStubs": { "limit": 40 @@ -105,13 +105,13 @@ "limit": 109 }, "reportUnknownMemberType": { - "limit": 38335 + "limit": 38332 }, "reportUnknownParameterType": { "limit": 19625 }, "reportUnknownVariableType": { - "limit": 29865 + "limit": 29861 }, "reportUnnecessaryCast": { "limit": 111 diff --git a/type-discipline-budget.json b/type-discipline-budget.json index 45d00a71d39..091c1f46e02 100644 --- a/type-discipline-budget.json +++ b/type-discipline-budget.json @@ -6,7 +6,7 @@ "limit": 26774 }, "LIT003": { - "limit": 262 + "limit": 261 }, "LIT004": { "limit": 40 @@ -27,10 +27,10 @@ "limit": 0 }, "LIT010": { - "limit": 16488 + "limit": 16486 }, "LIT011": { - "limit": 5523 + "limit": 5521 }, "LIT012": { "limit": 4495 From 9aeeca4ce3b8c9d6a9545ac65523c44bcd1f41db Mon Sep 17 00:00:00 2001 From: tin-berri Date: Wed, 2 Sep 2026 16:33:04 -0700 Subject: [PATCH 30/31] feat(router): add heuristic v2 complexity routing (#39276) * feat(router): add trained heuristic complexity routing * feat(router): expose heuristic v2 classifier * style(router): format heuristic v2 predictor --- .../complexity_router/README.md | 30 + .../artifacts/ultrafeedback_tiers.json | 4069 +++++++++++++++++ .../complexity_router/complexity_router.py | 31 + .../complexity_router/config.py | 21 +- .../complexity_router/tier_predictor.py | 156 + litellm/types/utils.py | 1 + pyproject.toml | 5 +- .../router_strategy/test_complexity_router.py | 70 +- .../test_complexity_tier_predictor.py | 91 + .../add_model/ClassificationMethodConfig.tsx | 16 + .../add_model/ComplexityRouterConfig.test.tsx | 25 + .../add_model/ComplexityRouterConfig.tsx | 19 +- .../add_model/HeuristicScoringConfig.test.tsx | 1 + .../build_complexity_router_config.test.ts | 21 +- .../LogDetailsDrawer/RoutingDecisionCard.tsx | 1 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 115 +- 16 files changed, 4653 insertions(+), 19 deletions(-) create mode 100644 litellm/router_strategy/complexity_router/artifacts/ultrafeedback_tiers.json create mode 100644 litellm/router_strategy/complexity_router/tier_predictor.py create mode 100644 tests/test_litellm/router_strategy/test_complexity_tier_predictor.py diff --git a/litellm/router_strategy/complexity_router/README.md b/litellm/router_strategy/complexity_router/README.md index bc8df67cc28..8e0cad39561 100644 --- a/litellm/router_strategy/complexity_router/README.md +++ b/litellm/router_strategy/complexity_router/README.md @@ -68,6 +68,36 @@ still resolve to a deployment in `model_list`; this configuration does not creat - abc ``` +### Heuristic v2 + +Set `classifier_type: heuristic_v2` to classify with the bundled calibrated +success-probability model instead of the hand-written weighted scorer + +```yaml +model_list: + - model_name: smart-router + litellm_params: + model: auto_router/complexity_router + complexity_router_config: + classifier_type: heuristic_v2 + tiers: + SIMPLE: luna + MEDIUM: terra + COMPLEX: sol + REASONING: sol-ultra +``` + +No classifier model call or per-model training data is required. The classifier +uses global tier quality, request-type quality, and similar-request cohorts from +the bundled UltraFeedback artifact. It estimates success at every tier, enforces +monotonic probabilities, and returns the first tier meeting the trained 0.75 +threshold. The existing complexity-router tier pool then selects and dispatches +a model from that tier + +Spend logs record `routing_decision.cause: heuristic_v2`, the detected request +type, and all four predicted probabilities. Existing `classifier_type: heuristic` +configurations keep the original weighted scorer unchanged + ### Renaming the tiers `tier_labels` puts your own vocabulary on the four tiers: diff --git a/litellm/router_strategy/complexity_router/artifacts/ultrafeedback_tiers.json b/litellm/router_strategy/complexity_router/artifacts/ultrafeedback_tiers.json new file mode 100644 index 00000000000..4fcb599907c --- /dev/null +++ b/litellm/router_strategy/complexity_router/artifacts/ultrafeedback_tiers.json @@ -0,0 +1,4069 @@ +{ + "schema_version": 1, + "global_statistics": [ + { + "tier": 1, + "successes": 36619.0, + "observations": 45504.0 + }, + { + "tier": 2, + "successes": 59797.0, + "observations": 70062.0 + }, + { + "tier": 3, + "successes": 48604.0, + "observations": 52245.0 + }, + { + "tier": 4, + "successes": 11393.0, + "observations": 11561.0 + } + ], + "domain_statistics": [ + { + "tier": 1, + "successes": 1592.0, + "observations": 2211.0, + "request_type": "analytical_reasoning" + }, + { + "tier": 2, + "successes": 2654.0, + "observations": 3374.0, + "request_type": "analytical_reasoning" + }, + { + "tier": 3, + "successes": 2243.0, + "observations": 2481.0, + "request_type": "analytical_reasoning" + }, + { + "tier": 4, + "successes": 538.0, + "observations": 546.0, + "request_type": "analytical_reasoning" + }, + { + "tier": 1, + "successes": 750.0, + "observations": 1015.0, + "request_type": "code_generation" + }, + { + "tier": 2, + "successes": 1271.0, + "observations": 1511.0, + "request_type": "code_generation" + }, + { + "tier": 3, + "successes": 1030.0, + "observations": 1111.0, + "request_type": "code_generation" + }, + { + "tier": 4, + "successes": 233.0, + "observations": 235.0, + "request_type": "code_generation" + }, + { + "tier": 1, + "successes": 243.0, + "observations": 277.0, + "request_type": "code_understanding" + }, + { + "tier": 2, + "successes": 385.0, + "observations": 425.0, + "request_type": "code_understanding" + }, + { + "tier": 3, + "successes": 322.0, + "observations": 334.0, + "request_type": "code_understanding" + }, + { + "tier": 4, + "successes": 74.0, + "observations": 76.0, + "request_type": "code_understanding" + }, + { + "tier": 1, + "successes": 2014.0, + "observations": 2170.0, + "request_type": "factual_lookup" + }, + { + "tier": 2, + "successes": 3120.0, + "observations": 3266.0, + "request_type": "factual_lookup" + }, + { + "tier": 3, + "successes": 2612.0, + "observations": 2670.0, + "request_type": "factual_lookup" + }, + { + "tier": 4, + "successes": 540.0, + "observations": 542.0, + "request_type": "factual_lookup" + }, + { + "tier": 1, + "successes": 30571.0, + "observations": 38161.0, + "request_type": "general" + }, + { + "tier": 2, + "successes": 50037.0, + "observations": 58821.0, + "request_type": "general" + }, + { + "tier": 3, + "successes": 40460.0, + "observations": 43618.0, + "request_type": "general" + }, + { + "tier": 4, + "successes": 9565.0, + "observations": 9716.0, + "request_type": "general" + }, + { + "tier": 1, + "successes": 159.0, + "observations": 170.0, + "request_type": "technical_design" + }, + { + "tier": 2, + "successes": 282.0, + "observations": 303.0, + "request_type": "technical_design" + }, + { + "tier": 3, + "successes": 231.0, + "observations": 236.0, + "request_type": "technical_design" + }, + { + "tier": 4, + "successes": 47.0, + "observations": 47.0, + "request_type": "technical_design" + }, + { + "tier": 1, + "successes": 1290.0, + "observations": 1500.0, + "request_type": "writing" + }, + { + "tier": 2, + "successes": 2048.0, + "observations": 2362.0, + "request_type": "writing" + }, + { + "tier": 3, + "successes": 1706.0, + "observations": 1795.0, + "request_type": "writing" + }, + { + "tier": 4, + "successes": 396.0, + "observations": 399.0, + "request_type": "writing" + } + ], + "cohort_statistics": [ + { + "tier": 1, + "successes": 272.0, + "observations": 372.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 420.0, + "observations": 519.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 341.0, + "observations": 381.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 82.0, + "observations": 84.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 0.0, + "observations": 1.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 0.0, + "observations": 2.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 11.0, + "observations": 18.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 33.0, + "observations": 45.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 28.0, + "observations": 34.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "analytical_reasoning|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 131.0, + "observations": 176.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 210.0, + "observations": 274.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 187.0, + "observations": 209.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 39.0, + "observations": 41.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 4.0, + "observations": 9.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 11.0, + "observations": 13.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 6.0, + "observations": 7.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 10.0, + "observations": 15.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 14.0, + "observations": 16.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 15.0, + "observations": 16.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "analytical_reasoning|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 15.0, + "observations": 20.0, + "cohort": "analytical_reasoning|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 33.0, + "observations": 38.0, + "cohort": "analytical_reasoning|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 28.0, + "observations": 29.0, + "cohort": "analytical_reasoning|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "analytical_reasoning|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 2.0, + "cohort": "analytical_reasoning|long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "analytical_reasoning|long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 26.0, + "observations": 34.0, + "cohort": "analytical_reasoning|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 39.0, + "observations": 50.0, + "cohort": "analytical_reasoning|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 42.0, + "observations": 45.0, + "cohort": "analytical_reasoning|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 11.0, + "observations": 11.0, + "cohort": "analytical_reasoning|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 3.0, + "cohort": "analytical_reasoning|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 4.0, + "observations": 4.0, + "cohort": "analytical_reasoning|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 452.0, + "observations": 634.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 745.0, + "observations": 962.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 634.0, + "observations": 691.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 151.0, + "observations": 153.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 3.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 4.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 10.0, + "observations": 20.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 30.0, + "observations": 44.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 12.0, + "observations": 15.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "analytical_reasoning|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 178.0, + "observations": 270.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 304.0, + "observations": 402.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 266.0, + "observations": 295.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 73.0, + "observations": 73.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 7.0, + "observations": 18.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 17.0, + "observations": 32.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 17.0, + "observations": 27.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 7.0, + "observations": 7.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 11.0, + "observations": 17.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 17.0, + "observations": 25.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 10.0, + "observations": 16.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 6.0, + "observations": 6.0, + "cohort": "analytical_reasoning|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 58.0, + "observations": 72.0, + "cohort": "analytical_reasoning|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 90.0, + "observations": 103.0, + "cohort": "analytical_reasoning|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 72.0, + "observations": 78.0, + "cohort": "analytical_reasoning|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 19.0, + "observations": 19.0, + "cohort": "analytical_reasoning|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "analytical_reasoning|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 37.0, + "observations": 56.0, + "cohort": "analytical_reasoning|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 95.0, + "observations": 110.0, + "cohort": "analytical_reasoning|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 73.0, + "observations": 82.0, + "cohort": "analytical_reasoning|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 20.0, + "observations": 20.0, + "cohort": "analytical_reasoning|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 69.0, + "observations": 83.0, + "cohort": "analytical_reasoning|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 102.0, + "observations": 110.0, + "cohort": "analytical_reasoning|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 93.0, + "observations": 98.0, + "cohort": "analytical_reasoning|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 17.0, + "observations": 17.0, + "cohort": "analytical_reasoning|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|short|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 6.0, + "observations": 6.0, + "cohort": "analytical_reasoning|short|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|short|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 21.0, + "observations": 32.0, + "cohort": "analytical_reasoning|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 45.0, + "observations": 59.0, + "cohort": "analytical_reasoning|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 45.0, + "observations": 48.0, + "cohort": "analytical_reasoning|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 9.0, + "observations": 9.0, + "cohort": "analytical_reasoning|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|short|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "analytical_reasoning|short|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|short|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 7.0, + "observations": 9.0, + "cohort": "analytical_reasoning|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 8.0, + "observations": 12.0, + "cohort": "analytical_reasoning|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 11.0, + "observations": 11.0, + "cohort": "analytical_reasoning|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 5.0, + "observations": 9.0, + "cohort": "analytical_reasoning|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 9.0, + "observations": 11.0, + "cohort": "analytical_reasoning|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 8.0, + "observations": 8.0, + "cohort": "analytical_reasoning|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 136.0, + "observations": 176.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 221.0, + "observations": 267.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 172.0, + "observations": 194.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 37.0, + "observations": 39.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 6.0, + "observations": 11.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 10.0, + "observations": 15.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 8.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "analytical_reasoning|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 68.0, + "observations": 84.0, + "cohort": "analytical_reasoning|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 103.0, + "observations": 136.0, + "cohort": "analytical_reasoning|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 113.0, + "observations": 119.0, + "cohort": "analytical_reasoning|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 21.0, + "observations": 21.0, + "cohort": "analytical_reasoning|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 3.0, + "cohort": "analytical_reasoning|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 4.0, + "observations": 4.0, + "cohort": "analytical_reasoning|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 13.0, + "observations": 16.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 27.0, + "observations": 34.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 15.0, + "observations": 16.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 6.0, + "observations": 6.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 4.0, + "observations": 5.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 4.0, + "observations": 7.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 3.0, + "observations": 3.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 32.0, + "observations": 39.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 48.0, + "observations": 63.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 35.0, + "observations": 40.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 14.0, + "observations": 14.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 3.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 2.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 2.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "analytical_reasoning|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 28.0, + "observations": 31.0, + "cohort": "code_generation|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 41.0, + "observations": 46.0, + "cohort": "code_generation|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 25.0, + "observations": 26.0, + "cohort": "code_generation|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "code_generation|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 6.0, + "observations": 9.0, + "cohort": "code_generation|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 20.0, + "observations": 23.0, + "cohort": "code_generation|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 16.0, + "observations": 16.0, + "cohort": "code_generation|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 4.0, + "observations": 4.0, + "cohort": "code_generation|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 46.0, + "observations": 60.0, + "cohort": "code_generation|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 72.0, + "observations": 91.0, + "cohort": "code_generation|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 59.0, + "observations": 64.0, + "cohort": "code_generation|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 17.0, + "observations": 17.0, + "cohort": "code_generation|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 32.0, + "observations": 49.0, + "cohort": "code_generation|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 60.0, + "observations": 76.0, + "cohort": "code_generation|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 48.0, + "observations": 52.0, + "cohort": "code_generation|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 15.0, + "observations": 15.0, + "cohort": "code_generation|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 93.0, + "observations": 121.0, + "cohort": "code_generation|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 149.0, + "observations": 170.0, + "cohort": "code_generation|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 146.0, + "observations": 151.0, + "cohort": "code_generation|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 26.0, + "observations": 26.0, + "cohort": "code_generation|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 12.0, + "observations": 16.0, + "cohort": "code_generation|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 22.0, + "observations": 25.0, + "cohort": "code_generation|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 13.0, + "observations": 15.0, + "cohort": "code_generation|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 4.0, + "observations": 4.0, + "cohort": "code_generation|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 196.0, + "observations": 268.0, + "cohort": "code_generation|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 312.0, + "observations": 370.0, + "cohort": "code_generation|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 234.0, + "observations": 253.0, + "cohort": "code_generation|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 56.0, + "observations": 57.0, + "cohort": "code_generation|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 0.0, + "observations": 1.0, + "cohort": "code_generation|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_generation|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_generation|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 40.0, + "observations": 59.0, + "cohort": "code_generation|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 73.0, + "observations": 92.0, + "cohort": "code_generation|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 78.0, + "observations": 83.0, + "cohort": "code_generation|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 14.0, + "observations": 14.0, + "cohort": "code_generation|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 3.0, + "cohort": "code_generation|medium|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_generation|medium|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 106.0, + "observations": 140.0, + "cohort": "code_generation|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 220.0, + "observations": 247.0, + "cohort": "code_generation|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 167.0, + "observations": 178.0, + "cohort": "code_generation|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 43.0, + "observations": 43.0, + "cohort": "code_generation|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 4.0, + "cohort": "code_generation|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 7.0, + "observations": 9.0, + "cohort": "code_generation|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 7.0, + "cohort": "code_generation|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 147.0, + "observations": 199.0, + "cohort": "code_generation|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 233.0, + "observations": 269.0, + "cohort": "code_generation|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 189.0, + "observations": 209.0, + "cohort": "code_generation|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 34.0, + "observations": 35.0, + "cohort": "code_generation|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 7.0, + "observations": 10.0, + "cohort": "code_generation|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 18.0, + "observations": 22.0, + "cohort": "code_generation|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 10.0, + "observations": 10.0, + "cohort": "code_generation|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_generation|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 4.0, + "cohort": "code_generation|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 5.0, + "observations": 8.0, + "cohort": "code_generation|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 8.0, + "cohort": "code_generation|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_generation|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 3.0, + "observations": 5.0, + "cohort": "code_generation|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 3.0, + "observations": 3.0, + "cohort": "code_generation|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_generation|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 7.0, + "observations": 9.0, + "cohort": "code_generation|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 7.0, + "observations": 8.0, + "cohort": "code_generation|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 9.0, + "observations": 9.0, + "cohort": "code_generation|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_generation|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 24.0, + "observations": 33.0, + "cohort": "code_generation|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 25.0, + "observations": 45.0, + "cohort": "code_generation|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 19.0, + "observations": 27.0, + "cohort": "code_generation|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 7.0, + "observations": 7.0, + "cohort": "code_generation|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_understanding|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_understanding|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 6.0, + "observations": 6.0, + "cohort": "code_understanding|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_understanding|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 5.0, + "cohort": "code_understanding|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 3.0, + "observations": 3.0, + "cohort": "code_understanding|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 3.0, + "cohort": "code_understanding|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 17.0, + "observations": 23.0, + "cohort": "code_understanding|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 27.0, + "observations": 37.0, + "cohort": "code_understanding|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 32.0, + "observations": 33.0, + "cohort": "code_understanding|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "code_understanding|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 24.0, + "observations": 28.0, + "cohort": "code_understanding|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 37.0, + "observations": 44.0, + "cohort": "code_understanding|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 33.0, + "observations": 33.0, + "cohort": "code_understanding|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 7.0, + "observations": 7.0, + "cohort": "code_understanding|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 8.0, + "observations": 9.0, + "cohort": "code_understanding|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 19.0, + "observations": 21.0, + "cohort": "code_understanding|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 10.0, + "observations": 10.0, + "cohort": "code_understanding|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 4.0, + "observations": 4.0, + "cohort": "code_understanding|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 3.0, + "cohort": "code_understanding|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 3.0, + "cohort": "code_understanding|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 73.0, + "observations": 76.0, + "cohort": "code_understanding|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 107.0, + "observations": 111.0, + "cohort": "code_understanding|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 87.0, + "observations": 87.0, + "cohort": "code_understanding|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 22.0, + "observations": 22.0, + "cohort": "code_understanding|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 32.0, + "observations": 33.0, + "cohort": "code_understanding|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 44.0, + "observations": 44.0, + "cohort": "code_understanding|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 44.0, + "observations": 47.0, + "cohort": "code_understanding|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 8.0, + "observations": 8.0, + "cohort": "code_understanding|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 0.0, + "observations": 1.0, + "cohort": "code_understanding|medium|code=1|math=1|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 0.0, + "observations": 2.0, + "cohort": "code_understanding|medium|code=1|math=1|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|medium|code=1|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 15.0, + "observations": 16.0, + "cohort": "code_understanding|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 23.0, + "observations": 23.0, + "cohort": "code_understanding|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 12.0, + "observations": 14.0, + "cohort": "code_understanding|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "code_understanding|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 3.0, + "cohort": "code_understanding|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 36.0, + "observations": 43.0, + "cohort": "code_understanding|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 71.0, + "observations": 74.0, + "cohort": "code_understanding|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 48.0, + "observations": 50.0, + "cohort": "code_understanding|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 12.0, + "observations": 13.0, + "cohort": "code_understanding|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 5.0, + "observations": 6.0, + "cohort": "code_understanding|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 10.0, + "observations": 10.0, + "cohort": "code_understanding|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 7.0, + "cohort": "code_understanding|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 0.0, + "observations": 1.0, + "cohort": "code_understanding|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "code_understanding|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 9.0, + "observations": 11.0, + "cohort": "code_understanding|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 18.0, + "observations": 23.0, + "cohort": "code_understanding|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 11.0, + "observations": 13.0, + "cohort": "code_understanding|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 3.0, + "cohort": "code_understanding|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 1.0, + "cohort": "code_understanding|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 13.0, + "observations": 17.0, + "cohort": "code_understanding|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 19.0, + "observations": 23.0, + "cohort": "code_understanding|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 27.0, + "observations": 28.0, + "cohort": "code_understanding|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 7.0, + "observations": 8.0, + "cohort": "code_understanding|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 49.0, + "observations": 54.0, + "cohort": "factual_lookup|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 64.0, + "observations": 75.0, + "cohort": "factual_lookup|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 80.0, + "observations": 80.0, + "cohort": "factual_lookup|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 15.0, + "observations": 15.0, + "cohort": "factual_lookup|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 9.0, + "observations": 9.0, + "cohort": "factual_lookup|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 12.0, + "observations": 14.0, + "cohort": "factual_lookup|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 9.0, + "observations": 10.0, + "cohort": "factual_lookup|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "factual_lookup|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 3.0, + "observations": 3.0, + "cohort": "factual_lookup|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 19.0, + "observations": 22.0, + "cohort": "factual_lookup|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 35.0, + "observations": 38.0, + "cohort": "factual_lookup|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 37.0, + "observations": 41.0, + "cohort": "factual_lookup|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 7.0, + "observations": 7.0, + "cohort": "factual_lookup|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 66.0, + "observations": 75.0, + "cohort": "factual_lookup|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 73.0, + "observations": 86.0, + "cohort": "factual_lookup|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 69.0, + "observations": 74.0, + "cohort": "factual_lookup|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 21.0, + "observations": 21.0, + "cohort": "factual_lookup|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "factual_lookup|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 174.0, + "observations": 181.0, + "cohort": "factual_lookup|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 204.0, + "observations": 215.0, + "cohort": "factual_lookup|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 206.0, + "observations": 210.0, + "cohort": "factual_lookup|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 42.0, + "observations": 42.0, + "cohort": "factual_lookup|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 31.0, + "observations": 37.0, + "cohort": "factual_lookup|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 41.0, + "observations": 48.0, + "cohort": "factual_lookup|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 44.0, + "observations": 45.0, + "cohort": "factual_lookup|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 10.0, + "observations": 10.0, + "cohort": "factual_lookup|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 0.0, + "observations": 1.0, + "cohort": "factual_lookup|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "factual_lookup|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 131.0, + "observations": 142.0, + "cohort": "factual_lookup|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 162.0, + "observations": 171.0, + "cohort": "factual_lookup|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 146.0, + "observations": 151.0, + "cohort": "factual_lookup|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 24.0, + "observations": 24.0, + "cohort": "factual_lookup|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 41.0, + "observations": 46.0, + "cohort": "factual_lookup|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 55.0, + "observations": 59.0, + "cohort": "factual_lookup|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 45.0, + "observations": 46.0, + "cohort": "factual_lookup|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 9.0, + "observations": 9.0, + "cohort": "factual_lookup|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1382.0, + "observations": 1473.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 2290.0, + "observations": 2348.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 1824.0, + "observations": 1853.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 368.0, + "observations": 370.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 5.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 12.0, + "observations": 13.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 7.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "factual_lookup|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 11.0, + "observations": 16.0, + "cohort": "factual_lookup|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 28.0, + "observations": 31.0, + "cohort": "factual_lookup|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 15.0, + "observations": 17.0, + "cohort": "factual_lookup|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 4.0, + "observations": 4.0, + "cohort": "factual_lookup|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 36.0, + "observations": 39.0, + "cohort": "factual_lookup|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 63.0, + "observations": 66.0, + "cohort": "factual_lookup|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 64.0, + "observations": 66.0, + "cohort": "factual_lookup|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 13.0, + "observations": 13.0, + "cohort": "factual_lookup|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 4.0, + "observations": 4.0, + "cohort": "factual_lookup|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 5.0, + "observations": 6.0, + "cohort": "factual_lookup|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 4.0, + "observations": 5.0, + "cohort": "factual_lookup|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 27.0, + "observations": 31.0, + "cohort": "factual_lookup|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 29.0, + "observations": 37.0, + "cohort": "factual_lookup|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 26.0, + "observations": 29.0, + "cohort": "factual_lookup|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 11.0, + "observations": 11.0, + "cohort": "factual_lookup|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 0.0, + "observations": 2.0, + "cohort": "factual_lookup|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 7.0, + "observations": 8.0, + "cohort": "factual_lookup|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 17.0, + "observations": 20.0, + "cohort": "factual_lookup|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 14.0, + "observations": 14.0, + "cohort": "factual_lookup|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "factual_lookup|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 4.0, + "observations": 4.0, + "cohort": "factual_lookup|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 6.0, + "observations": 6.0, + "cohort": "factual_lookup|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 5.0, + "observations": 5.0, + "cohort": "factual_lookup|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "factual_lookup|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 17.0, + "observations": 21.0, + "cohort": "factual_lookup|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 19.0, + "observations": 25.0, + "cohort": "factual_lookup|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 13.0, + "observations": 13.0, + "cohort": "factual_lookup|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "factual_lookup|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2828.0, + "observations": 3552.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 4621.0, + "observations": 5551.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 3628.0, + "observations": 3931.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 911.0, + "observations": 922.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 127.0, + "observations": 406.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 250.0, + "observations": 583.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 219.0, + "observations": 396.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 98.0, + "observations": 107.0, + "cohort": "general|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 94.0, + "observations": 127.0, + "cohort": "general|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 168.0, + "observations": 196.0, + "cohort": "general|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 135.0, + "observations": 146.0, + "cohort": "general|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 34.0, + "observations": 35.0, + "cohort": "general|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 912.0, + "observations": 1219.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 1417.0, + "observations": 1736.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 1163.0, + "observations": 1258.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 278.0, + "observations": 283.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 33.0, + "observations": 122.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 41.0, + "observations": 150.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 61.0, + "observations": 96.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 24.0, + "observations": 24.0, + "cohort": "general|long|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 16.0, + "observations": 24.0, + "cohort": "general|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 24.0, + "observations": 34.0, + "cohort": "general|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 18.0, + "observations": 21.0, + "cohort": "general|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "general|long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 271.0, + "observations": 348.0, + "cohort": "general|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 488.0, + "observations": 555.0, + "cohort": "general|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 369.0, + "observations": 386.0, + "cohort": "general|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 67.0, + "observations": 71.0, + "cohort": "general|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 3.0, + "cohort": "general|long|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 0.0, + "observations": 3.0, + "cohort": "general|long|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "general|long|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 9.0, + "observations": 9.0, + "cohort": "general|long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 14.0, + "observations": 17.0, + "cohort": "general|long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 9.0, + "observations": 10.0, + "cohort": "general|long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 4.0, + "observations": 4.0, + "cohort": "general|long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 340.0, + "observations": 426.0, + "cohort": "general|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 544.0, + "observations": 635.0, + "cohort": "general|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 472.0, + "observations": 491.0, + "cohort": "general|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 120.0, + "observations": 120.0, + "cohort": "general|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|long|code=1|math=1|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "general|long|code=1|math=1|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|long|code=1|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 0.0, + "observations": 2.0, + "cohort": "general|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 0.0, + "observations": 1.0, + "cohort": "general|long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 8332.0, + "observations": 10133.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 13225.0, + "observations": 15509.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 10614.0, + "observations": 11347.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2499.0, + "observations": 2531.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 437.0, + "observations": 1294.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 863.0, + "observations": 1932.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 738.0, + "observations": 1269.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 296.0, + "observations": 325.0, + "cohort": "general|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 117.0, + "observations": 173.0, + "cohort": "general|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 183.0, + "observations": 252.0, + "cohort": "general|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 171.0, + "observations": 191.0, + "cohort": "general|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 47.0, + "observations": 52.0, + "cohort": "general|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 927.0, + "observations": 1353.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 1550.0, + "observations": 2023.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 1273.0, + "observations": 1430.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 348.0, + "observations": 354.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 60.0, + "observations": 255.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 143.0, + "observations": 377.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 145.0, + "observations": 243.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 58.0, + "observations": 61.0, + "cohort": "general|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 27.0, + "observations": 47.0, + "cohort": "general|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 30.0, + "observations": 48.0, + "cohort": "general|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 30.0, + "observations": 36.0, + "cohort": "general|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "general|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 1119.0, + "observations": 1348.0, + "cohort": "general|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 1816.0, + "observations": 2024.0, + "cohort": "general|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 1472.0, + "observations": 1552.0, + "cohort": "general|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 301.0, + "observations": 304.0, + "cohort": "general|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|medium|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|medium|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "general|medium|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 8.0, + "observations": 10.0, + "cohort": "general|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 9.0, + "observations": 12.0, + "cohort": "general|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 8.0, + "observations": 9.0, + "cohort": "general|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|medium|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 349.0, + "observations": 439.0, + "cohort": "general|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 549.0, + "observations": 622.0, + "cohort": "general|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 495.0, + "observations": 526.0, + "cohort": "general|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 111.0, + "observations": 113.0, + "cohort": "general|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|medium|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 4.0, + "observations": 6.0, + "cohort": "general|medium|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 4.0, + "observations": 4.0, + "cohort": "general|medium|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|medium|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 11635.0, + "observations": 12910.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 19249.0, + "observations": 20591.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 15415.0, + "observations": 15867.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 3362.0, + "observations": 3392.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 68.0, + "observations": 155.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 113.0, + "observations": 202.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 97.0, + "observations": 124.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 31.0, + "observations": 31.0, + "cohort": "general|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 6.0, + "observations": 11.0, + "cohort": "general|short|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 20.0, + "observations": 21.0, + "cohort": "general|short|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 23.0, + "observations": 24.0, + "cohort": "general|short|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 4.0, + "observations": 4.0, + "cohort": "general|short|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 199.0, + "observations": 270.0, + "cohort": "general|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 334.0, + "observations": 414.0, + "cohort": "general|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 309.0, + "observations": 337.0, + "cohort": "general|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 75.0, + "observations": 75.0, + "cohort": "general|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 5.0, + "cohort": "general|short|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 5.0, + "cohort": "general|short|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "general|short|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|short|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "general|short|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "general|short|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 629.0, + "observations": 778.0, + "cohort": "general|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 1131.0, + "observations": 1286.0, + "cohort": "general|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 919.0, + "observations": 989.0, + "cohort": "general|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 227.0, + "observations": 227.0, + "cohort": "general|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 4.0, + "cohort": "general|short|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 3.0, + "observations": 3.0, + "cohort": "general|short|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 4.0, + "observations": 5.0, + "cohort": "general|short|code=1|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "general|short|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "general|short|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 28.0, + "observations": 37.0, + "cohort": "general|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 54.0, + "observations": 64.0, + "cohort": "general|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 41.0, + "observations": 46.0, + "cohort": "general|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 9.0, + "observations": 9.0, + "cohort": "general|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1066.0, + "observations": 1437.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 1639.0, + "observations": 2007.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 1397.0, + "observations": 1507.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 333.0, + "observations": 341.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 6.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 4.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 3.0, + "observations": 3.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "general|very_long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 65.0, + "observations": 89.0, + "cohort": "general|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 119.0, + "observations": 139.0, + "cohort": "general|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 98.0, + "observations": 103.0, + "cohort": "general|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 21.0, + "observations": 21.0, + "cohort": "general|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 403.0, + "observations": 546.0, + "cohort": "general|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 700.0, + "observations": 875.0, + "cohort": "general|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 556.0, + "observations": 612.0, + "cohort": "general|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 133.0, + "observations": 135.0, + "cohort": "general|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 17.0, + "observations": 27.0, + "cohort": "general|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 19.0, + "observations": 33.0, + "cohort": "general|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 25.0, + "observations": 27.0, + "cohort": "general|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 9.0, + "observations": 9.0, + "cohort": "general|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 210.0, + "observations": 280.0, + "cohort": "general|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 307.0, + "observations": 378.0, + "cohort": "general|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 258.0, + "observations": 289.0, + "cohort": "general|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 64.0, + "observations": 65.0, + "cohort": "general|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 17.0, + "observations": 23.0, + "cohort": "general|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 31.0, + "observations": 39.0, + "cohort": "general|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 16.0, + "observations": 19.0, + "cohort": "general|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "general|very_long|code=1|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 202.0, + "observations": 282.0, + "cohort": "general|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 361.0, + "observations": 478.0, + "cohort": "general|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 262.0, + "observations": 310.0, + "cohort": "general|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 82.0, + "observations": 82.0, + "cohort": "general|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 4.0, + "observations": 5.0, + "cohort": "general|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 9.0, + "observations": 11.0, + "cohort": "general|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 4.0, + "observations": 4.0, + "cohort": "general|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 19.0, + "observations": 21.0, + "cohort": "technical_design|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 23.0, + "observations": 25.0, + "cohort": "technical_design|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 28.0, + "observations": 29.0, + "cohort": "technical_design|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "technical_design|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 8.0, + "observations": 9.0, + "cohort": "technical_design|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 9.0, + "observations": 9.0, + "cohort": "technical_design|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 14.0, + "observations": 15.0, + "cohort": "technical_design|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 24.0, + "observations": 25.0, + "cohort": "technical_design|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 18.0, + "observations": 18.0, + "cohort": "technical_design|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 44.0, + "observations": 45.0, + "cohort": "technical_design|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 84.0, + "observations": 86.0, + "cohort": "technical_design|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 74.0, + "observations": 74.0, + "cohort": "technical_design|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 15.0, + "observations": 15.0, + "cohort": "technical_design|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 6.0, + "observations": 6.0, + "cohort": "technical_design|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 23.0, + "observations": 24.0, + "cohort": "technical_design|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 47.0, + "observations": 53.0, + "cohort": "technical_design|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 28.0, + "observations": 29.0, + "cohort": "technical_design|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 10.0, + "observations": 10.0, + "cohort": "technical_design|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 4.0, + "observations": 4.0, + "cohort": "technical_design|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 7.0, + "observations": 7.0, + "cohort": "technical_design|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 8.0, + "cohort": "technical_design|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 13.0, + "observations": 13.0, + "cohort": "technical_design|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 31.0, + "observations": 31.0, + "cohort": "technical_design|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 23.0, + "observations": 23.0, + "cohort": "technical_design|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 5.0, + "observations": 5.0, + "cohort": "technical_design|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 3.0, + "observations": 4.0, + "cohort": "technical_design|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 10.0, + "observations": 15.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 26.0, + "observations": 30.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 19.0, + "observations": 21.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 5.0, + "observations": 6.0, + "cohort": "technical_design|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 6.0, + "observations": 7.0, + "cohort": "technical_design|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 6.0, + "observations": 6.0, + "cohort": "technical_design|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 4.0, + "observations": 6.0, + "cohort": "technical_design|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 8.0, + "observations": 8.0, + "cohort": "technical_design|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 5.0, + "observations": 6.0, + "cohort": "technical_design|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 5.0, + "observations": 5.0, + "cohort": "technical_design|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "technical_design|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "technical_design|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 3.0, + "observations": 3.0, + "cohort": "technical_design|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 176.0, + "observations": 226.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 267.0, + "observations": 329.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 229.0, + "observations": 245.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 51.0, + "observations": 52.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 0.0, + "observations": 2.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 2.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 3.0, + "observations": 3.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|long|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 5.0, + "cohort": "writing|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 4.0, + "observations": 5.0, + "cohort": "writing|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 36.0, + "observations": 47.0, + "cohort": "writing|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 53.0, + "observations": 59.0, + "cohort": "writing|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 47.0, + "observations": 51.0, + "cohort": "writing|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 14.0, + "observations": 15.0, + "cohort": "writing|long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 17.0, + "observations": 21.0, + "cohort": "writing|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 27.0, + "observations": 32.0, + "cohort": "writing|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 24.0, + "observations": 24.0, + "cohort": "writing|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "writing|long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 2.0, + "cohort": "writing|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 2.0, + "cohort": "writing|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 6.0, + "observations": 6.0, + "cohort": "writing|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 541.0, + "observations": 598.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 892.0, + "observations": 997.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 728.0, + "observations": 756.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 164.0, + "observations": 165.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 7.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 7.0, + "observations": 15.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 10.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 4.0, + "observations": 4.0, + "cohort": "writing|medium|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|medium|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 25.0, + "observations": 35.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 41.0, + "observations": 63.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 39.0, + "observations": 46.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 12.0, + "observations": 12.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 2.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 4.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|medium|code=0|math=1|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|medium|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 24.0, + "observations": 27.0, + "cohort": "writing|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 49.0, + "observations": 56.0, + "cohort": "writing|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 46.0, + "observations": 49.0, + "cohort": "writing|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 8.0, + "observations": 8.0, + "cohort": "writing|medium|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 11.0, + "observations": 13.0, + "cohort": "writing|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 13.0, + "observations": 13.0, + "cohort": "writing|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 10.0, + "observations": 10.0, + "cohort": "writing|medium|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 327.0, + "observations": 353.0, + "cohort": "writing|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 508.0, + "observations": 555.0, + "cohort": "writing|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 415.0, + "observations": 432.0, + "cohort": "writing|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 92.0, + "observations": 92.0, + "cohort": "writing|short|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|short|code=0|math=0|mc=0|intl=1" + }, + { + "tier": 1, + "successes": 4.0, + "observations": 4.0, + "cohort": "writing|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 7.0, + "observations": 7.0, + "cohort": "writing|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 3.0, + "observations": 3.0, + "cohort": "writing|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|short|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 3.0, + "cohort": "writing|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 8.0, + "observations": 9.0, + "cohort": "writing|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 4.0, + "observations": 4.0, + "cohort": "writing|short|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|short|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 63.0, + "observations": 85.0, + "cohort": "writing|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 111.0, + "observations": 139.0, + "cohort": "writing|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 95.0, + "observations": 99.0, + "cohort": "writing|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 29.0, + "observations": 29.0, + "cohort": "writing|very_long|code=0|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 3.0, + "observations": 5.0, + "cohort": "writing|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 2.0, + "cohort": "writing|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 4.0, + "observations": 4.0, + "cohort": "writing|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 4, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|very_long|code=0|math=0|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 26.0, + "observations": 36.0, + "cohort": "writing|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 30.0, + "observations": 44.0, + "cohort": "writing|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 23.0, + "observations": 28.0, + "cohort": "writing|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 8.0, + "observations": 8.0, + "cohort": "writing|very_long|code=0|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 1.0, + "observations": 1.0, + "cohort": "writing|very_long|code=0|math=1|mc=1|intl=0" + }, + { + "tier": 1, + "successes": 14.0, + "observations": 14.0, + "cohort": "writing|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 14.0, + "observations": 16.0, + "cohort": "writing|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 6.0, + "observations": 7.0, + "cohort": "writing|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 4, + "successes": 3.0, + "observations": 3.0, + "cohort": "writing|very_long|code=1|math=0|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 5.0, + "observations": 5.0, + "cohort": "writing|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 2, + "successes": 7.0, + "observations": 8.0, + "cohort": "writing|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 3, + "successes": 7.0, + "observations": 7.0, + "cohort": "writing|very_long|code=1|math=1|mc=0|intl=0" + }, + { + "tier": 1, + "successes": 0.0, + "observations": 1.0, + "cohort": "writing|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 2, + "successes": 2.0, + "observations": 2.0, + "cohort": "writing|very_long|code=1|math=1|mc=1|intl=0" + }, + { + "tier": 3, + "successes": 0.0, + "observations": 1.0, + "cohort": "writing|very_long|code=1|math=1|mc=1|intl=0" + } + ], + "domain_prior_mass": 200.0, + "cohort_prior_mass": 20.0, + "routing_threshold": 0.75, + "datasets": [ + { + "name": "openbmb/UltraFeedback", + "url": "https://huggingface.co/datasets/openbmb/UltraFeedback", + "license": "MIT", + "rows": 255864, + "success_definition": "UltraFeedback overall_score >= 4" + } + ], + "success_definition": "UltraFeedback overall_score >= 4", + "split_method": "sha256(prompt): 70% train, 15% validation, 15% test" +} diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index 577cee0920d..a123a75dd81 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -33,6 +33,11 @@ from litellm.litellm_core_utils.internal_call_metadata import forwarded_internal from litellm.litellm_core_utils.prompt_templates.common_utils import request_contains_image_content from litellm.litellm_core_utils.sensitive_data_masker import mask_credentials_in_payload from litellm.llms.base_llm.base_utils import type_to_response_format_param +from litellm.router_strategy.adaptive_router.classifier import classify_prompt +from litellm.router_strategy.complexity_router.tier_predictor import ( + TierSuccessPredictor, + resolve_tier_artifact, +) from litellm.types.utils import ( AUTOROUTER_CLASSIFIER_CALL_ORIGIN, ModelResponse, @@ -790,6 +795,7 @@ class ClassificationOutcome(NamedTuple): signals: tuple[str, ...] cause: Literal[ "heuristic_scorer", + "heuristic_v2", "reasoning_override", "llm_classifier", "heuristic_first_short_circuit", @@ -978,6 +984,11 @@ class ComplexityRouter(CustomLogger): if llm_classifier_configured else None ) + self._tier_success_predictor: TierSuccessPredictor | None = ( + TierSuccessPredictor(resolve_tier_artifact(self.config.heuristic_v2_artifact)) + if self.config.classifier_type == "heuristic_v2" + else None + ) verbose_router_logger.debug("ComplexityRouter initialized for %s with tiers: %s", model_name, self.config.tiers) @@ -1350,6 +1361,8 @@ class ComplexityRouter(CustomLogger): custom tier set, and classifier_fallback otherwise decides between the heuristic scorer and default_model. The outcome's `cause` reports which path actually ran. """ + if self.config.classifier_type == "heuristic_v2": + return self._classify_with_heuristic_v2(prompt) if self.config.classifier_type == "custom": return await self._classify_with_plugin(prompt, system_prompt, request_kwargs, raw_messages) if self.config.classifier_type == "heuristic_first" and self.config.classifier_llm_config is not None: @@ -1359,6 +1372,24 @@ class ComplexityRouter(CustomLogger): return ClassificationOutcome(tier=tier, score=score, signals=signals, cause=cause) return await self._llm_classifier_outcome(prompt, system_prompt, request_kwargs, messages) + def _classify_with_heuristic_v2(self, prompt: str) -> ClassificationOutcome: + predictor: Final = self._tier_success_predictor + if predictor is None: + raise ValueError("heuristic v2 predictor is not configured") + request_type: Final = classify_prompt(prompt) + prediction: Final = predictor.predict(prompt, request_type) + tier: Final = TIER_SEVERITY_ORDER[prediction.required_tier - 1] + probability_signals: Final = tuple( + f"tier-probability:{candidate.value.lower()}={prediction.probabilities[index]:.6f}" + for index, candidate in enumerate(TIER_SEVERITY_ORDER, start=1) + ) + return ClassificationOutcome( + tier=tier, + score=None, + signals=(f"request-type:{request_type.value}", *probability_signals), + cause="heuristic_v2", + ) + async def _classify_heuristic_first( self, prompt: str, diff --git a/litellm/router_strategy/complexity_router/config.py b/litellm/router_strategy/complexity_router/config.py index 70aeecb31c6..fdf2a3a0b39 100644 --- a/litellm/router_strategy/complexity_router/config.py +++ b/litellm/router_strategy/complexity_router/config.py @@ -14,6 +14,8 @@ from pydantic import BaseModel, ConfigDict, Field, SkipValidation, field_seriali from litellm.types.router import AdaptiveRouterWeights, ClassifierPlugin, RoutingPlugin +from .tier_predictor import TrainedTierArtifact + class ComplexityTier(str, Enum): """Complexity tiers for routing decisions.""" @@ -625,12 +627,19 @@ class ComplexityRouterConfig(BaseModel): ) # Classifier strategy - classifier_type: Literal["heuristic", "llm", "custom", "heuristic_first"] = Field( + classifier_type: Literal["heuristic", "heuristic_v2", "llm", "custom", "heuristic_first"] = Field( default="heuristic", description=( - "Classification strategy: local regex/keyword scoring, an LLM call, a custom classifier " - "plugin, or 'heuristic_first', which scores locally and only pays for the LLM classifier " - "when the local scorer does not confidently land a cheap tier" + "Classification strategy: local regex/keyword scoring, the bundled trained four-tier heuristic, " + "an LLM call, a custom classifier plugin, or 'heuristic_first', which scores locally and only pays " + "for the LLM classifier when the local scorer does not confidently land a cheap tier" + ), + ) + heuristic_v2_artifact: TrainedTierArtifact | Literal["ultrafeedback"] = Field( + default="ultrafeedback", + description=( + "Success-probability artifact used by classifier_type 'heuristic_v2'. The bundled " + "UltraFeedback artifact is selected by default; an inline trained artifact may replace it" ), ) classifier_llm_config: ClassifierLLMConfig | None = Field( @@ -1248,10 +1257,10 @@ class ComplexityRouterConfig(BaseModel): ) if duplicated: raise ValueError(f"tier_definitions names must be unique (case-insensitive): {', '.join(duplicated)}") - if self.classifier_type in ("heuristic", "heuristic_first"): + if self.classifier_type in ("heuristic", "heuristic_v2", "heuristic_first"): raise ValueError( "tier_definitions requires classifier_type 'llm' or 'custom': the heuristic scorer only " - "produces the built-in tiers" + "produces the four built-in tiers, as does heuristic_v2" ) conflicts: Final = self._tier_definition_conflicts() if conflicts: diff --git a/litellm/router_strategy/complexity_router/tier_predictor.py b/litellm/router_strategy/complexity_router/tier_predictor.py new file mode 100644 index 00000000000..764f6e6ad56 --- /dev/null +++ b/litellm/router_strategy/complexity_router/tier_predictor.py @@ -0,0 +1,156 @@ +from __future__ import annotations + +import re +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path +from types import MappingProxyType +from typing import Final, Literal + +from pydantic import BaseModel, Field, model_validator + +from litellm.types.router import RequestType + + +class TierGlobalStatistic(BaseModel): + tier: int = Field(ge=1, le=4) + successes: float = Field(ge=0.0) + observations: float = Field(gt=0.0) + + @model_validator(mode="after") + def _successes_do_not_exceed_observations(self) -> TierGlobalStatistic: + if self.successes > self.observations: + raise ValueError("successes cannot exceed observations") + return self + + +class TierDomainStatistic(TierGlobalStatistic): + request_type: RequestType + + +class TierCohortStatistic(TierGlobalStatistic): + cohort: str = Field(min_length=1) + + +class TierDataset(BaseModel): + name: str = Field(min_length=1) + url: str = Field(min_length=1) + license: str = Field(min_length=1) + rows: int = Field(gt=0) + success_definition: str = Field(default="quality score meets the dataset success threshold", min_length=1) + + +class TrainedTierArtifact(BaseModel): + schema_version: Literal[1] = 1 + global_statistics: tuple[TierGlobalStatistic, ...] + domain_statistics: tuple[TierDomainStatistic, ...] = () + cohort_statistics: tuple[TierCohortStatistic, ...] = () + domain_prior_mass: float = Field(default=200.0, gt=0.0) + cohort_prior_mass: float = Field(default=20.0, gt=0.0) + routing_threshold: float = Field(default=0.75, ge=0.0, le=1.0) + datasets: tuple[TierDataset, ...] = () + success_definition: str = Field(default="quality score meets the dataset success threshold", min_length=1) + split_method: str = Field(default="sha256(prompt): 70% train, 15% validation, 15% test", min_length=1) + + @model_validator(mode="after") + def _statistics_are_unique(self) -> TrainedTierArtifact: + global_tiers: Final = tuple(stat.tier for stat in self.global_statistics) + if frozenset(global_tiers) != frozenset((1, 2, 3, 4)) or len(global_tiers) != 4: + raise ValueError("global statistics must contain each tier exactly once") + domain_keys: Final = tuple((stat.request_type, stat.tier) for stat in self.domain_statistics) + if len(domain_keys) != len(frozenset(domain_keys)): + raise ValueError("domain statistics must contain unique request_type and tier pairs") + cohort_keys: Final = tuple((stat.cohort, stat.tier) for stat in self.cohort_statistics) + if len(cohort_keys) != len(frozenset(cohort_keys)): + raise ValueError("cohort statistics must contain unique cohort and tier pairs") + return self + + +_CODE_PATTERN: Final = re.compile( + r"```|\b(def|class|function|python|javascript|typescript|sql|code)\b", + re.IGNORECASE, +) +_MATH_PATTERN: Final = re.compile( + r"\b(solve|calculate|equation|probability|theorem|proof|integral)\b|[$=]", + re.IGNORECASE, +) +_MULTIPLE_CHOICE_PATTERN: Final = re.compile(r"(?:^|\s)[A-D][.)]\s") +_TIERS: Final = (1, 2, 3, 4) +_BUILTIN_ARTIFACTS: Final = MappingProxyType({"ultrafeedback": "ultrafeedback_tiers.json"}) + + +def resolve_tier_artifact(artifact: TrainedTierArtifact | str) -> TrainedTierArtifact: + if isinstance(artifact, TrainedTierArtifact): + return artifact + filename: Final = _BUILTIN_ARTIFACTS.get(artifact) + if filename is None: + raise ValueError(f"unknown complexity router tier artifact: {artifact}") + path: Final = Path(__file__).with_name("artifacts") / filename + return TrainedTierArtifact.model_validate_json(path.read_text()) + + +def similarity_cohort(prompt: str, request_type: RequestType) -> str: + length: Final = len(prompt) + length_bucket: Final = ( + "short" if length < 200 else "medium" if length < 800 else "long" if length < 2000 else "very_long" + ) + code: Final = int(bool(_CODE_PATTERN.search(prompt))) + math: Final = int(bool(_MATH_PATTERN.search(prompt))) + multiple_choice: Final = int(bool(_MULTIPLE_CHOICE_PATTERN.search(prompt))) + non_ascii: Final = int(sum(ord(character) > 127 for character in prompt) / max(1, length) > 0.1) + return f"{request_type.value}|{length_bucket}|code={code}|math={math}|mc={multiple_choice}|intl={non_ascii}" + + +@dataclass(frozen=True, slots=True) +class TierPrediction: + probabilities: Mapping[int, float] + required_tier: int + + +class TierSuccessPredictor: + def __init__(self, artifact: TrainedTierArtifact) -> None: + self._artifact = artifact + self._global: Mapping[int, TierGlobalStatistic] = MappingProxyType( + {stat.tier: stat for stat in artifact.global_statistics} + ) + self._domain: Mapping[tuple[RequestType, int], TierDomainStatistic] = MappingProxyType( + {(stat.request_type, stat.tier): stat for stat in artifact.domain_statistics} + ) + self._cohort: Mapping[tuple[str, int], TierCohortStatistic] = MappingProxyType( + {(stat.cohort, stat.tier): stat for stat in artifact.cohort_statistics} + ) + + @property + def routing_threshold(self) -> float: + return self._artifact.routing_threshold + + def predict(self, prompt: str, request_type: RequestType) -> TierPrediction: + cohort: Final = similarity_cohort(prompt, request_type) + raw: Final = tuple(self._probability(tier, request_type, cohort) for tier in _TIERS) + monotonic: Final = tuple(max(raw[:index]) for index in range(1, len(raw) + 1)) + probabilities: Final[Mapping[int, float]] = MappingProxyType( + {int(tier): probability for tier, probability in zip(_TIERS, monotonic)} + ) + required_tier: Final = next( + (tier for tier in _TIERS if probabilities[tier] >= self._artifact.routing_threshold), + 4, + ) + return TierPrediction(probabilities=probabilities, required_tier=required_tier) + + def _probability(self, tier: int, request_type: RequestType, cohort: str) -> float: + global_stat: Final = self._global[tier] + global_mean: Final = (global_stat.successes + 1.0) / (global_stat.observations + 2.0) + domain_stat: Final = self._domain.get((request_type, tier)) + domain_mean: Final = self._posterior_mean(domain_stat, self._artifact.domain_prior_mass, global_mean) + cohort_stat: Final = self._cohort.get((cohort, tier)) + return self._posterior_mean(cohort_stat, self._artifact.cohort_prior_mass, domain_mean) + + @staticmethod + def _posterior_mean( + statistic: TierGlobalStatistic | None, + prior_mass: float, + prior_mean: float, + ) -> float: + if statistic is None: + return prior_mean + return (statistic.successes + prior_mass * prior_mean) / (statistic.observations + prior_mass) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 5783a39b30c..09c01873a9d 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2839,6 +2839,7 @@ class StandardLoggingRoutingDecisionTierBoundaries(TypedDict): RoutingDecisionCause = Literal[ "heuristic_scorer", + "heuristic_v2", # The scorer found 2+ reasoning markers and forced REASONING regardless of score. # A distinct cause rather than a marker inside `signals`, because it is the fact # that tells a reader the score did NOT choose the tier; encoding it as free text diff --git a/pyproject.toml b/pyproject.toml index 60162544612..d0e5723d1cb 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -278,7 +278,10 @@ bindings = "pyo3" features = ["extension-module"] profile = "release" editable-profile = "dev" -include = ["litellm/proxy/_experimental/out/**"] +include = [ + "litellm/proxy/_experimental/out/**", + "litellm/router_strategy/complexity_router/artifacts/*.json", +] exclude = [ "litellm/proxy/enterprise", "litellm/proxy/enterprise/**", diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index 1ec8be88c9b..82659571de4 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -12,7 +12,6 @@ from unittest.mock import AsyncMock, MagicMock, patch import pytest from pydantic import ValidationError - import litellm from litellm import Router from litellm._logging import verbose_router_logger @@ -34,10 +33,14 @@ from litellm.router_strategy.complexity_router.config import ( DEFAULT_CLASSIFIER_CONTEXT_WINDOW_SIZE, DEFAULT_COMPLEXITY_CONFIG, DEFAULT_TECHNICAL_KEYWORDS, + ClassificationRubric, ClassifierLLMConfig, ComplexityRouterConfig, ComplexityTier, - ClassificationRubric, +) +from litellm.router_strategy.complexity_router.tier_predictor import ( + TierGlobalStatistic, + TrainedTierArtifact, ) from litellm.types.router import ( Deployment, @@ -46,6 +49,16 @@ from litellm.types.router import ( ) +def _heuristic_v2_artifact() -> TrainedTierArtifact: + return TrainedTierArtifact( + global_statistics=tuple( + TierGlobalStatistic(tier=tier, successes=successes, observations=100) + for tier, successes in enumerate((10, 20, 90, 99), start=1) + ), + routing_threshold=0.8, + ) + + @pytest.fixture def mock_router_instance(): """Create a mock LiteLLM Router instance.""" @@ -1696,6 +1709,59 @@ class TestLLMClassifier: assert outcome.cause == "heuristic_scorer" assert outcome.score is not None + @pytest.mark.asyncio + async def test_heuristic_v2_routes_directly_to_predicted_builtin_tier(self, mock_router_instance): + router = ComplexityRouter( + model_name="tier-router", + litellm_router_instance=mock_router_instance, + complexity_router_config={ + "classifier_type": "heuristic_v2", + "heuristic_v2_artifact": _heuristic_v2_artifact(), + "tiers": { + "SIMPLE": "simple-model", + "MEDIUM": "medium-model", + "COMPLEX": "complex-model", + "REASONING": "reasoning-model", + }, + }, + ) + + response = await router.async_pre_routing_hook( + model="tier-router", + request_kwargs={}, + messages=[{"role": "user", "content": "Handle this new request"}], + ) + + assert response is not None + assert response.model == "complex-model" + assert response.routing_decision["tier"] == "COMPLEX" + assert response.routing_decision["cause"] == "heuristic_v2" + assert response.routing_decision["signals"] == [ + "request-type:general", + "tier-probability:simple=0.107843", + "tier-probability:medium=0.205882", + "tier-probability:complex=0.892157", + "tier-probability:reasoning=0.980392", + ] + + def test_heuristic_v2_needs_no_classifier_model(self): + config = ComplexityRouterConfig(classifier_type="heuristic_v2") + + assert config.classifier_llm_config is None + assert config.heuristic_v2_artifact == "ultrafeedback" + + def test_heuristic_v2_rejects_custom_tier_definitions(self): + with pytest.raises(ValidationError, match="as does heuristic_v2"): + ComplexityRouterConfig( + classifier_type="heuristic_v2", + tier_definitions=( + {"name": "low", "description": "easy work"}, + {"name": "high", "description": "hard work"}, + ), + tiers={"low": "cheap", "high": "expensive"}, + fallback_tier="high", + ) + @pytest.mark.asyncio async def test_aclassify_llm_success_routes_by_llm_verdict(self, llm_complexity_router, mock_router_instance): """A well-formed structured LLM response should decide the tier directly. diff --git a/tests/test_litellm/router_strategy/test_complexity_tier_predictor.py b/tests/test_litellm/router_strategy/test_complexity_tier_predictor.py new file mode 100644 index 00000000000..5bbe0fb5669 --- /dev/null +++ b/tests/test_litellm/router_strategy/test_complexity_tier_predictor.py @@ -0,0 +1,91 @@ +from typing import Final + +import pytest + +from litellm.router_strategy.complexity_router.tier_predictor import ( + TierCohortStatistic, + TierDomainStatistic, + TierGlobalStatistic, + TierSuccessPredictor, + TrainedTierArtifact, + resolve_tier_artifact, + similarity_cohort, +) +from litellm.types.router import RequestType + + +def _artifact( + global_successes: tuple[float, float, float, float] = (4.0, 5.0, 6.0, 7.0), + threshold: float = 0.75, + domain_statistics: tuple[TierDomainStatistic, ...] = (), + cohort_statistics: tuple[TierCohortStatistic, ...] = (), +) -> TrainedTierArtifact: + return TrainedTierArtifact( + global_statistics=tuple( + TierGlobalStatistic(tier=tier, successes=successes, observations=10.0) + for tier, successes in enumerate(global_successes, start=1) + ), + domain_statistics=domain_statistics, + cohort_statistics=cohort_statistics, + domain_prior_mass=10.0, + cohort_prior_mass=10.0, + routing_threshold=threshold, + ) + + +def test_predictions_are_monotonic_across_tiers() -> None: + predictor: Final = TierSuccessPredictor(_artifact(global_successes=(9.0, 2.0, 7.0, 6.0))) + + prediction: Final = predictor.predict("hello", RequestType.GENERAL) + + probabilities: Final = tuple(prediction.probabilities.values()) + assert probabilities == tuple(sorted(probabilities)) + + +def test_domain_and_cohort_statistics_back_off_hierarchically() -> None: + matching_cohort: Final = similarity_cohort("hello", RequestType.GENERAL) + artifact: Final = _artifact( + global_successes=(1.0, 5.0, 6.0, 7.0), + domain_statistics=( + TierDomainStatistic( + tier=1, + request_type=RequestType.GENERAL, + successes=10.0, + observations=10.0, + ), + ), + cohort_statistics=( + TierCohortStatistic( + tier=1, + cohort=matching_cohort, + successes=0.0, + observations=10.0, + ), + ), + ) + predictor: Final = TierSuccessPredictor(artifact) + + cohort_probability: Final = predictor.predict("hello", RequestType.GENERAL).probabilities[1] + domain_probability: Final = predictor.predict("hello " * 100, RequestType.GENERAL).probabilities[1] + global_probability: Final = predictor.predict("hello", RequestType.WRITING).probabilities[1] + + assert cohort_probability == pytest.approx(7.0 / 24.0) + assert domain_probability == pytest.approx(7.0 / 12.0) + assert global_probability == pytest.approx(1.0 / 6.0) + + +def test_selects_first_tier_above_probability_threshold() -> None: + predictor: Final = TierSuccessPredictor(_artifact(global_successes=(4.0, 6.0, 8.0, 9.0), threshold=0.7)) + + prediction: Final = predictor.predict("hello", RequestType.GENERAL) + + assert prediction.required_tier == 3 + + +def test_builtin_ultrafeedback_artifact_is_loadable() -> None: + artifact: Final = resolve_tier_artifact("ultrafeedback") + + assert artifact.routing_threshold == 0.75 + assert artifact.domain_prior_mass == 200.0 + assert artifact.cohort_prior_mass == 20.0 + assert artifact.datasets[0].license == "MIT" diff --git a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx index 96c93306611..c2ed4b1f55a 100644 --- a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx +++ b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx @@ -43,6 +43,10 @@ const DEFAULT_SCORING_EXPLANATION = "The router scores each request across 7 dimensions: token count, code presence, reasoning markers, technical " + "terms, simple indicators, multi-step patterns, and question complexity. The weighted score determines the tier:"; +const HEURISTIC_V2_EXPLANATION = + "The router estimates success probability for all four tiers with the bundled calibrated model, then selects " + + "the first tier that meets its trained threshold. It runs locally with no classifier API call."; + const CLASSIFIER_TIMEOUT_ID = "classifier-timeout-ms"; const CLASSIFIER_CONTEXT_WINDOW_SIZE_ID = "classifier-context-window-size"; const CLASSIFIER_CONTEXT_BUDGET_CHARS_ID = "classifier-context-budget-chars"; @@ -62,6 +66,7 @@ const CUSTOM_PROMPT_WITH_DEFAULT_MODEL_FALLBACK = * at all, so the panel must not keep implying a score is involved on either router. */ const scoringExplanation = (value: ComplexityRouterConfigValue): string => { + if (value.classifier_type === "heuristic_v2") return HEURISTIC_V2_EXPLANATION; const usesCustomPrompt = usesLlmClassifier(value.classifier_type) && Boolean(value.classifier_llm_config?.system_prompt?.trim()); if (!usesCustomPrompt) return DEFAULT_SCORING_EXPLANATION; @@ -179,6 +184,17 @@ const ClassifierTypeRadios: React.FC<{ + + +