mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
refactor: route hosted_vllm Anthropic passthrough through ProviderConfigManager
- Move _should_skip_anthropic_translation to hosted_vllm/messages/transformation.py - Register HostedVLLMAnthropicMessagesConfig in ProviderConfigManager instead of hard-coding a provider branch in the Anthropic messages handler - Pass litellm_params to get_provider_anthropic_messages_config so the flag check can happen inside the abstraction used by all other providers - Pin ProviderConfigManager mock in test_flag_disabled_uses_translation so the test is resilient to future hosted_vllm config registrations Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
parent
b7cfd2d026
commit
c0f7ba3094
4 changed files with 40 additions and 29 deletions
|
|
@ -7,7 +7,6 @@
|
|||
|
||||
import asyncio
|
||||
import contextvars
|
||||
import os
|
||||
from functools import partial
|
||||
from typing import Any, AsyncIterator, Coroutine, Dict, List, Optional, Union, cast
|
||||
|
||||
|
|
@ -22,7 +21,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
|
|||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
|
||||
from litellm.llms.hosted_vllm.messages.transformation import (
|
||||
HostedVLLMAnthropicMessagesConfig,
|
||||
_should_skip_anthropic_translation, # re-exported for backwards compat
|
||||
)
|
||||
from litellm.types.llms.anthropic_messages.anthropic_request import AnthropicMetadata
|
||||
from litellm.types.llms.anthropic_messages.anthropic_response import (
|
||||
|
|
@ -54,23 +53,6 @@ def _should_route_to_responses_api(custom_llm_provider: Optional[str]) -> bool:
|
|||
return custom_llm_provider in _RESPONSES_API_PROVIDERS
|
||||
|
||||
|
||||
def _should_skip_anthropic_translation(litellm_params: GenericLiteLLMParams) -> bool:
|
||||
"""Return True when Anthropic→OpenAI translation should be bypassed for hosted_vllm.
|
||||
|
||||
Checked in priority order:
|
||||
1. Per-deployment ``disable_anthropic_translation`` in litellm_params
|
||||
2. Global env var ``DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION``
|
||||
"""
|
||||
param_flag = litellm_params.get("disable_anthropic_translation")
|
||||
if param_flag is not None:
|
||||
return bool(param_flag)
|
||||
return os.environ.get("DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION", "").lower() in (
|
||||
"true",
|
||||
"1",
|
||||
"yes",
|
||||
)
|
||||
|
||||
|
||||
####### ENVIRONMENT VARIABLES ###################
|
||||
# Initialize any necessary instances or variables here
|
||||
base_llm_http_handler = BaseLLMHTTPHandler()
|
||||
|
|
@ -431,17 +413,14 @@ def anthropic_messages_handler(
|
|||
|
||||
anthropic_messages_provider_config: Optional[BaseAnthropicMessagesConfig] = None
|
||||
|
||||
if custom_llm_provider == "hosted_vllm" and _should_skip_anthropic_translation(
|
||||
litellm_params
|
||||
):
|
||||
anthropic_messages_provider_config = HostedVLLMAnthropicMessagesConfig()
|
||||
elif custom_llm_provider is not None and custom_llm_provider in [
|
||||
if custom_llm_provider is not None and custom_llm_provider in [
|
||||
provider.value for provider in LlmProviders
|
||||
]:
|
||||
anthropic_messages_provider_config = (
|
||||
ProviderConfigManager.get_provider_anthropic_messages_config(
|
||||
model=model,
|
||||
provider=litellm.LlmProviders(custom_llm_provider),
|
||||
litellm_params=litellm_params,
|
||||
)
|
||||
)
|
||||
if anthropic_messages_provider_config is None:
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ Anthropic→OpenAI chat/completions translation and POSTs the original Anthropic
|
|||
directly to `{api_base}/v1/messages`.
|
||||
"""
|
||||
|
||||
import os
|
||||
from typing import Any, AsyncIterator, Dict, List, Optional, Tuple
|
||||
|
||||
import httpx
|
||||
|
|
@ -21,6 +22,23 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
|
|||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
|
||||
def _should_skip_anthropic_translation(litellm_params: GenericLiteLLMParams) -> bool:
|
||||
"""Return True when Anthropic→OpenAI translation should be bypassed for hosted_vllm.
|
||||
|
||||
Checked in priority order:
|
||||
1. Per-deployment ``disable_anthropic_translation`` in litellm_params
|
||||
2. Global env var ``DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION``
|
||||
"""
|
||||
param_flag = litellm_params.get("disable_anthropic_translation")
|
||||
if param_flag is not None:
|
||||
return bool(param_flag)
|
||||
return os.environ.get("DISABLE_HOSTED_VLLM_ANTHROPIC_TRANSLATION", "").lower() in (
|
||||
"true",
|
||||
"1",
|
||||
"yes",
|
||||
)
|
||||
|
||||
|
||||
class HostedVLLMAnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
||||
"""
|
||||
Sends Anthropic-format /v1/messages requests directly to a vLLM instance.
|
||||
|
|
|
|||
|
|
@ -8499,7 +8499,19 @@ class ProviderConfigManager:
|
|||
def get_provider_anthropic_messages_config(
|
||||
model: str,
|
||||
provider: LlmProviders,
|
||||
litellm_params: Optional[Any] = None,
|
||||
) -> Optional[BaseAnthropicMessagesConfig]:
|
||||
if provider == litellm.LlmProviders.HOSTED_VLLM:
|
||||
from litellm.llms.hosted_vllm.messages.transformation import (
|
||||
HostedVLLMAnthropicMessagesConfig,
|
||||
_should_skip_anthropic_translation,
|
||||
)
|
||||
|
||||
if litellm_params is not None and _should_skip_anthropic_translation(
|
||||
litellm_params
|
||||
):
|
||||
return HostedVLLMAnthropicMessagesConfig()
|
||||
return None
|
||||
return ProviderConfigManager._get_provider_anthropic_messages_config_cached(
|
||||
model=model, provider=provider
|
||||
)
|
||||
|
|
|
|||
|
|
@ -17,8 +17,6 @@ import pytest
|
|||
import litellm
|
||||
from litellm.llms.hosted_vllm.messages.transformation import (
|
||||
HostedVLLMAnthropicMessagesConfig,
|
||||
)
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
|
||||
_should_skip_anthropic_translation,
|
||||
)
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
|
@ -204,11 +202,13 @@ class TestHandlerRouting:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_flag_enabled_uses_passthrough_config(self):
|
||||
"""When disable_anthropic_translation=True, the handler must call
|
||||
base_llm_http_handler.anthropic_messages_handler (native path) instead of
|
||||
"""When disable_anthropic_translation=True, ProviderConfigManager returns
|
||||
HostedVLLMAnthropicMessagesConfig and the native path is used instead of
|
||||
LiteLLMMessagesToCompletionTransformationHandler."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler as h
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
passthrough_config = HostedVLLMAnthropicMessagesConfig()
|
||||
mock_response = MagicMock()
|
||||
mock_response.id = "msg_test"
|
||||
|
||||
|
|
@ -219,6 +219,7 @@ class TestHandlerRouting:
|
|||
"anthropic_messages_handler",
|
||||
) as mock_translate,
|
||||
patch("litellm.get_llm_provider", return_value=("qwen36-27b-fp8", "hosted_vllm", "key", "http://vllm/v1")),
|
||||
patch.object(ProviderConfigManager, "get_provider_anthropic_messages_config", return_value=passthrough_config),
|
||||
):
|
||||
h.anthropic_messages_handler(
|
||||
max_tokens=100,
|
||||
|
|
@ -232,7 +233,6 @@ class TestHandlerRouting:
|
|||
mock_native.assert_called_once()
|
||||
mock_translate.assert_not_called()
|
||||
|
||||
# Confirm HostedVLLMAnthropicMessagesConfig was passed in
|
||||
call_kwargs = mock_native.call_args.kwargs
|
||||
assert isinstance(
|
||||
call_kwargs.get("anthropic_messages_provider_config"),
|
||||
|
|
@ -244,6 +244,7 @@ class TestHandlerRouting:
|
|||
"""When disable_anthropic_translation is absent (default), the handler must
|
||||
route to LiteLLMMessagesToCompletionTransformationHandler."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler as h
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
mock_response = MagicMock()
|
||||
|
||||
|
|
@ -254,6 +255,7 @@ class TestHandlerRouting:
|
|||
return_value=mock_response,
|
||||
) as mock_translate,
|
||||
patch("litellm.get_llm_provider", return_value=("qwen36-27b-fp8", "hosted_vllm", "key", "http://vllm/v1")),
|
||||
patch.object(ProviderConfigManager, "get_provider_anthropic_messages_config", return_value=None),
|
||||
):
|
||||
h.anthropic_messages_handler(
|
||||
max_tokens=100,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue