refactor: route hosted_vllm Anthropic passthrough through ProviderConfigManager

- Move _should_skip_anthropic_translation to hosted_vllm/messages/transformation.py
- Register HostedVLLMAnthropicMessagesConfig in ProviderConfigManager instead of
  hard-coding a provider branch in the Anthropic messages handler
- Pass litellm_params to get_provider_anthropic_messages_config so the flag check
  can happen inside the abstraction used by all other providers
- Pin ProviderConfigManager mock in test_flag_disabled_uses_translation so the test
  is resilient to future hosted_vllm config registrations

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
Denis Khachyan 2026-05-24 11:36:32 +03:00
parent b7cfd2d026
commit c0f7ba3094
4 changed files with 40 additions and 29 deletions

View file

@ -7,7 +7,6 @@
import asyncio
import contextvars
import os
from functools import partial
from typing import Any, AsyncIterator, Coroutine, Dict, List, Optional, Union, cast
@ -22,7 +21,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
from litellm.llms.hosted_vllm.messages.transformation import (
HostedVLLMAnthropicMessagesConfig,
_should_skip_anthropic_translation, # re-exported for backwards compat
)
from litellm.types.llms.anthropic_messages.anthropic_request import AnthropicMetadata
from litellm.types.llms.anthropic_messages.anthropic_response import (
@ -54,23 +53,6 @@ def _should_route_to_responses_api(custom_llm_provider: Optional[str]) -> bool:
return custom_llm_provider in _RESPONSES_API_PROVIDERS
def _should_skip_anthropic_translation(litellm_params: GenericLiteLLMParams) -> bool:
"""Return True when Anthropic→OpenAI translation should be bypassed for hosted_vllm.
Checked in priority order:
1. Per-deployment ``disable_anthropic_translation`` in litellm_params
2. Global env var ``DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION``
"""
param_flag = litellm_params.get("disable_anthropic_translation")
if param_flag is not None:
return bool(param_flag)
return os.environ.get("DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION", "").lower() in (
"true",
"1",
"yes",
)
####### ENVIRONMENT VARIABLES ###################
# Initialize any necessary instances or variables here
base_llm_http_handler = BaseLLMHTTPHandler()
@ -431,17 +413,14 @@ def anthropic_messages_handler(
anthropic_messages_provider_config: Optional[BaseAnthropicMessagesConfig] = None
if custom_llm_provider == "hosted_vllm" and _should_skip_anthropic_translation(
litellm_params
):
anthropic_messages_provider_config = HostedVLLMAnthropicMessagesConfig()
elif custom_llm_provider is not None and custom_llm_provider in [
if custom_llm_provider is not None and custom_llm_provider in [
provider.value for provider in LlmProviders
]:
anthropic_messages_provider_config = (
ProviderConfigManager.get_provider_anthropic_messages_config(
model=model,
provider=litellm.LlmProviders(custom_llm_provider),
litellm_params=litellm_params,
)
)
if anthropic_messages_provider_config is None:

View file

@ -7,6 +7,7 @@ Anthropic→OpenAI chat/completions translation and POSTs the original Anthropic
directly to `{api_base}/v1/messages`.
"""
import os
from typing import Any, AsyncIterator, Dict, List, Optional, Tuple
import httpx
@ -21,6 +22,23 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
from litellm.types.router import GenericLiteLLMParams
def _should_skip_anthropic_translation(litellm_params: GenericLiteLLMParams) -> bool:
"""Return True when Anthropic→OpenAI translation should be bypassed for hosted_vllm.
Checked in priority order:
1. Per-deployment ``disable_anthropic_translation`` in litellm_params
2. Global env var ``DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION``
"""
param_flag = litellm_params.get("disable_anthropic_translation")
if param_flag is not None:
return bool(param_flag)
return os.environ.get("DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION", "").lower() in (
"true",
"1",
"yes",
)
class HostedVLLMAnthropicMessagesConfig(BaseAnthropicMessagesConfig):
"""
Sends Anthropic-format /v1/messages requests directly to a vLLM instance.

View file

@ -8499,7 +8499,19 @@ class ProviderConfigManager:
def get_provider_anthropic_messages_config(
model: str,
provider: LlmProviders,
litellm_params: Optional[Any] = None,
) -> Optional[BaseAnthropicMessagesConfig]:
if provider == litellm.LlmProviders.HOSTED_VLLM:
from litellm.llms.hosted_vllm.messages.transformation import (
HostedVLLMAnthropicMessagesConfig,
_should_skip_anthropic_translation,
)
if litellm_params is not None and _should_skip_anthropic_translation(
litellm_params
):
return HostedVLLMAnthropicMessagesConfig()
return None
return ProviderConfigManager._get_provider_anthropic_messages_config_cached(
model=model, provider=provider
)

View file

@ -17,8 +17,6 @@ import pytest
import litellm
from litellm.llms.hosted_vllm.messages.transformation import (
HostedVLLMAnthropicMessagesConfig,
)
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
_should_skip_anthropic_translation,
)
from litellm.types.router import GenericLiteLLMParams
@ -204,11 +202,13 @@ class TestHandlerRouting:
@pytest.mark.asyncio
async def test_flag_enabled_uses_passthrough_config(self):
"""When disable_anthropic_translation=True, the handler must call
base_llm_http_handler.anthropic_messages_handler (native path) instead of
"""When disable_anthropic_translation=True, ProviderConfigManager returns
HostedVLLMAnthropicMessagesConfig and the native path is used instead of
LiteLLMMessagesToCompletionTransformationHandler."""
from litellm.llms.anthropic.experimental_pass_through.messages import handler as h
from litellm.utils import ProviderConfigManager
passthrough_config = HostedVLLMAnthropicMessagesConfig()
mock_response = MagicMock()
mock_response.id = "msg_test"
@ -219,6 +219,7 @@ class TestHandlerRouting:
"anthropic_messages_handler",
) as mock_translate,
patch("litellm.get_llm_provider", return_value=("qwen36-27b-fp8", "hosted_vllm", "key", "http://vllm/v1")),
patch.object(ProviderConfigManager, "get_provider_anthropic_messages_config", return_value=passthrough_config),
):
h.anthropic_messages_handler(
max_tokens=100,
@ -232,7 +233,6 @@ class TestHandlerRouting:
mock_native.assert_called_once()
mock_translate.assert_not_called()
# Confirm HostedVLLMAnthropicMessagesConfig was passed in
call_kwargs = mock_native.call_args.kwargs
assert isinstance(
call_kwargs.get("anthropic_messages_provider_config"),
@ -244,6 +244,7 @@ class TestHandlerRouting:
"""When disable_anthropic_translation is absent (default), the handler must
route to LiteLLMMessagesToCompletionTransformationHandler."""
from litellm.llms.anthropic.experimental_pass_through.messages import handler as h
from litellm.utils import ProviderConfigManager
mock_response = MagicMock()
@ -254,6 +255,7 @@ class TestHandlerRouting:
return_value=mock_response,
) as mock_translate,
patch("litellm.get_llm_provider", return_value=("qwen36-27b-fp8", "hosted_vllm", "key", "http://vllm/v1")),
patch.object(ProviderConfigManager, "get_provider_anthropic_messages_config", return_value=None),
):
h.anthropic_messages_handler(
max_tokens=100,