From c0f7ba3094ea98ebd25d9b848a2e6d467aa90ede Mon Sep 17 00:00:00 2001 From: Denis Khachyan Date: Sun, 24 May 2026 11:36:32 +0300 Subject: [PATCH] refactor: route hosted_vllm Anthropic passthrough through ProviderConfigManager - Move _should_skip_anthropic_translation to hosted_vllm/messages/transformation.py - Register HostedVLLMAnthropicMessagesConfig in ProviderConfigManager instead of hard-coding a provider branch in the Anthropic messages handler - Pass litellm_params to get_provider_anthropic_messages_config so the flag check can happen inside the abstraction used by all other providers - Pin ProviderConfigManager mock in test_flag_disabled_uses_translation so the test is resilient to future hosted_vllm config registrations Co-Authored-By: Claude Opus 4.7 --- .../messages/handler.py | 27 +++---------------- .../hosted_vllm/messages/transformation.py | 18 +++++++++++++ litellm/utils.py | 12 +++++++++ .../hosted_vllm/test_anthropic_passthrough.py | 12 +++++---- 4 files changed, 40 insertions(+), 29 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 3ae30b05c8e..5fc14a8cdd0 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -7,7 +7,6 @@ import asyncio import contextvars -import os from functools import partial from typing import Any, AsyncIterator, Coroutine, Dict, List, Optional, Union, cast @@ -22,7 +21,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import ( from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler from litellm.llms.hosted_vllm.messages.transformation import ( - HostedVLLMAnthropicMessagesConfig, + _should_skip_anthropic_translation, # re-exported for backwards compat ) from litellm.types.llms.anthropic_messages.anthropic_request import AnthropicMetadata from litellm.types.llms.anthropic_messages.anthropic_response import ( @@ -54,23 +53,6 @@ def _should_route_to_responses_api(custom_llm_provider: Optional[str]) -> bool: return custom_llm_provider in _RESPONSES_API_PROVIDERS -def _should_skip_anthropic_translation(litellm_params: GenericLiteLLMParams) -> bool: - """Return True when Anthropic→OpenAI translation should be bypassed for hosted_vllm. - - Checked in priority order: - 1. Per-deployment ``disable_anthropic_translation`` in litellm_params - 2. Global env var ``DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION`` - """ - param_flag = litellm_params.get("disable_anthropic_translation") - if param_flag is not None: - return bool(param_flag) - return os.environ.get("DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION", "").lower() in ( - "true", - "1", - "yes", - ) - - ####### ENVIRONMENT VARIABLES ################### # Initialize any necessary instances or variables here base_llm_http_handler = BaseLLMHTTPHandler() @@ -431,17 +413,14 @@ def anthropic_messages_handler( anthropic_messages_provider_config: Optional[BaseAnthropicMessagesConfig] = None - if custom_llm_provider == "hosted_vllm" and _should_skip_anthropic_translation( - litellm_params - ): - anthropic_messages_provider_config = HostedVLLMAnthropicMessagesConfig() - elif custom_llm_provider is not None and custom_llm_provider in [ + if custom_llm_provider is not None and custom_llm_provider in [ provider.value for provider in LlmProviders ]: anthropic_messages_provider_config = ( ProviderConfigManager.get_provider_anthropic_messages_config( model=model, provider=litellm.LlmProviders(custom_llm_provider), + litellm_params=litellm_params, ) ) if anthropic_messages_provider_config is None: diff --git a/litellm/llms/hosted_vllm/messages/transformation.py b/litellm/llms/hosted_vllm/messages/transformation.py index 5c382db7159..c93888b73aa 100644 --- a/litellm/llms/hosted_vllm/messages/transformation.py +++ b/litellm/llms/hosted_vllm/messages/transformation.py @@ -7,6 +7,7 @@ Anthropic→OpenAI chat/completions translation and POSTs the original Anthropic directly to `{api_base}/v1/messages`. """ +import os from typing import Any, AsyncIterator, Dict, List, Optional, Tuple import httpx @@ -21,6 +22,23 @@ from litellm.types.llms.anthropic_messages.anthropic_response import ( from litellm.types.router import GenericLiteLLMParams +def _should_skip_anthropic_translation(litellm_params: GenericLiteLLMParams) -> bool: + """Return True when Anthropic→OpenAI translation should be bypassed for hosted_vllm. + + Checked in priority order: + 1. Per-deployment ``disable_anthropic_translation`` in litellm_params + 2. Global env var ``DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION`` + """ + param_flag = litellm_params.get("disable_anthropic_translation") + if param_flag is not None: + return bool(param_flag) + return os.environ.get("DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION", "").lower() in ( + "true", + "1", + "yes", + ) + + class HostedVLLMAnthropicMessagesConfig(BaseAnthropicMessagesConfig): """ Sends Anthropic-format /v1/messages requests directly to a vLLM instance. diff --git a/litellm/utils.py b/litellm/utils.py index 001c89fee4c..e051e724fc6 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8499,7 +8499,19 @@ class ProviderConfigManager: def get_provider_anthropic_messages_config( model: str, provider: LlmProviders, + litellm_params: Optional[Any] = None, ) -> Optional[BaseAnthropicMessagesConfig]: + if provider == litellm.LlmProviders.HOSTED_VLLM: + from litellm.llms.hosted_vllm.messages.transformation import ( + HostedVLLMAnthropicMessagesConfig, + _should_skip_anthropic_translation, + ) + + if litellm_params is not None and _should_skip_anthropic_translation( + litellm_params + ): + return HostedVLLMAnthropicMessagesConfig() + return None return ProviderConfigManager._get_provider_anthropic_messages_config_cached( model=model, provider=provider ) diff --git a/tests/litellm/llms/hosted_vllm/test_anthropic_passthrough.py b/tests/litellm/llms/hosted_vllm/test_anthropic_passthrough.py index 9e3e14d3031..ecc1766b33f 100644 --- a/tests/litellm/llms/hosted_vllm/test_anthropic_passthrough.py +++ b/tests/litellm/llms/hosted_vllm/test_anthropic_passthrough.py @@ -17,8 +17,6 @@ import pytest import litellm from litellm.llms.hosted_vllm.messages.transformation import ( HostedVLLMAnthropicMessagesConfig, -) -from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( _should_skip_anthropic_translation, ) from litellm.types.router import GenericLiteLLMParams @@ -204,11 +202,13 @@ class TestHandlerRouting: @pytest.mark.asyncio async def test_flag_enabled_uses_passthrough_config(self): - """When disable_anthropic_translation=True, the handler must call - base_llm_http_handler.anthropic_messages_handler (native path) instead of + """When disable_anthropic_translation=True, ProviderConfigManager returns + HostedVLLMAnthropicMessagesConfig and the native path is used instead of LiteLLMMessagesToCompletionTransformationHandler.""" from litellm.llms.anthropic.experimental_pass_through.messages import handler as h + from litellm.utils import ProviderConfigManager + passthrough_config = HostedVLLMAnthropicMessagesConfig() mock_response = MagicMock() mock_response.id = "msg_test" @@ -219,6 +219,7 @@ class TestHandlerRouting: "anthropic_messages_handler", ) as mock_translate, patch("litellm.get_llm_provider", return_value=("qwen36-27b-fp8", "hosted_vllm", "key", "http://vllm/v1")), + patch.object(ProviderConfigManager, "get_provider_anthropic_messages_config", return_value=passthrough_config), ): h.anthropic_messages_handler( max_tokens=100, @@ -232,7 +233,6 @@ class TestHandlerRouting: mock_native.assert_called_once() mock_translate.assert_not_called() - # Confirm HostedVLLMAnthropicMessagesConfig was passed in call_kwargs = mock_native.call_args.kwargs assert isinstance( call_kwargs.get("anthropic_messages_provider_config"), @@ -244,6 +244,7 @@ class TestHandlerRouting: """When disable_anthropic_translation is absent (default), the handler must route to LiteLLMMessagesToCompletionTransformationHandler.""" from litellm.llms.anthropic.experimental_pass_through.messages import handler as h + from litellm.utils import ProviderConfigManager mock_response = MagicMock() @@ -254,6 +255,7 @@ class TestHandlerRouting: return_value=mock_response, ) as mock_translate, patch("litellm.get_llm_provider", return_value=("qwen36-27b-fp8", "hosted_vllm", "key", "http://vllm/v1")), + patch.object(ProviderConfigManager, "get_provider_anthropic_messages_config", return_value=None), ): h.anthropic_messages_handler( max_tokens=100,