fix(health): probe Bedrock Mantle Claude deployments over the Anthropic Messages API (#44419)

* fix(health): probe Bedrock Mantle Claude deployments over the Anthropic Messages API

Bedrock Mantle serves Claude ids only on /anthropic/v1/messages, but health
checks probed every chat-mode deployment over /v1/chat/completions, so a
bedrock_mantle Claude deployment showed unhealthy while real /v1/messages
traffic to it succeeded

Add an anthropic_messages health check mode and make it the default for
bedrock_mantle Claude models. An explicit model_info.mode still wins, and
/health/test_connection and the Add Model form accept the new mode

* fix(health): resolve the test connection mode from the deployment when the request omits it

The Admin UI model page sent the mode /model/info had filled in from the cost
map back as the probe mode, so Test Connection on a Bedrock Mantle Claude
deployment still went over chat completions. The page now forwards only the
row's id, and /health/test_connection resolves a missing mode the way /health
does: the stored model_info.mode, then the mode the provider requires, then the
cost map.

* fix(health): resolve an omitted ahealth_check mode the way the proxy does

* fix(health): test connection honors a stored mode only for the stored model and rejects a non-string mode

A request that selects a stored deployment and sends a different litellm_params.model now resolves the probe mode from that model instead of the stored model_info.mode. A litellm_params.mode that is not a string answers 400 instead of 500. The Bedrock Mantle rule that Claude models are probed over the Messages API moves into the provider package.

* fix(health): shape test connection probe params for the model the request probes

A request that selects a stored deployment by id and overrides the model
resolved its probe mode from the overridden model but still injected
max_tokens from the stored mode, so an embedding override of an
anthropic_messages deployment failed with a Mistral 422 extra_forbidden

* fix(health): report an early ahealth_check failure as itself, not as a missing mode

With the mode resolved automatically when the caller omits it, a failure
before that resolution (no model, a non-string model, a provider that does
not resolve) was wrapped as "Missing mode", a hint that pointed at the wrong
fix and dropped raw_request_typed_dict from the result. Every failure now
returns the same shape.

---------

Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-03 17:05:47 -07:00 • committed by GitHub
parent 4d30f8c59b
commit f0eda6d2a6
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
14 changed files with 596 additions and 43 deletions

View file

@ -24,6 +24,32 @@ IMAGE_EDIT_HEALTH_CHECK_PROMPT: Final = (
"Add a small yellow star in the top right corner of this simple drawing of a blue circle on a white background"
)
ANTHROPIC_MESSAGES_HEALTH_CHECK_MAX_TOKENS: Final = 16
def native_health_check_mode(model: str, custom_llm_provider: str | None) -> Literal["anthropic_messages"] | None:
if custom_llm_provider != "bedrock_mantle":
return None
from litellm.llms.bedrock_mantle.common_utils import mantle_health_check_mode
return mantle_health_check_mode(model)
def _cost_map_mode(model: str) -> str | None:
import litellm
from litellm.litellm_core_utils.health_check_utils import OPTIONAL_STR
return OPTIONAL_STR.validate_python(litellm.model_cost.get(model, {}).get("mode"))
def default_health_check_mode(requested_model: str, model: str, custom_llm_provider: str) -> str:
return (
native_health_check_mode(model=model, custom_llm_provider=custom_llm_provider)
or _cost_map_mode(requested_model)
or _cost_map_mode(model)
or "chat"
)
def get_image_file_for_health_check() -> bytes:
"""Return the image used for health checks."""
@ -167,6 +193,7 @@ class HealthCheckHelpers:
"realtime",
"batch",
"responses",
"anthropic_messages",
"ocr",
"evaluation",
],
@ -254,6 +281,13 @@ class HealthCheckHelpers:
**_filter_model_params(model_params=model_params),
input=prompt or "test",
),
"anthropic_messages": lambda: litellm.anthropic_messages(
**{
"max_tokens": ANTHROPIC_MESSAGES_HEALTH_CHECK_MAX_TOKENS,
"messages": [{"role": "user", "content": prompt or "test"}],
**model_params,
}
),
"ocr": lambda: litellm.aocr(
**_filter_model_params(model_params=model_params),
document=_ocr_health_check_document(model=model, custom_llm_provider=custom_llm_provider),

View file

@ -9,6 +9,7 @@ from pydantic import TypeAdapter
from litellm.types.decisions import DecisionsCallParams
DECISIONS_CALL_PARAMS: Final[TypeAdapter[DecisionsCallParams]] = TypeAdapter(DecisionsCallParams)
OPTIONAL_STR: Final[TypeAdapter[str | None]] = TypeAdapter(str | None)
def _filter_model_params(model_params: dict) -> dict:

View file

@ -14,7 +14,7 @@ global state.
import re
from collections.abc import Mapping
from typing import Final
from typing import Final, Literal
from botocore.exceptions import (
CredentialRetrievalError,
@ -131,6 +131,10 @@ def is_mantle_claude_model(model: str) -> bool:
return "claude" in model.lower()
def mantle_health_check_mode(model: str) -> Literal["anthropic_messages"] | None:
return "anthropic_messages" if is_mantle_claude_model(model) else None
def mantle_supports_responses(model: str | None, model_cost: dict) -> bool:
"""Whether a Bedrock Mantle model can serve the native Responses API.

View file

@ -8725,7 +8725,7 @@ def speech(
async def ahealth_check(
model_params: dict,
mode: str | None = "chat",
mode: str | None = None,
prompt: str | None = None,
input: list | None = None,
):
@ -8740,7 +8740,8 @@ async def ahealth_check(
}
"""
from litellm.litellm_core_utils.cached_imports import get_litellm_logging_class
from litellm.litellm_core_utils.health_check_helpers import HealthCheckHelpers
from litellm.litellm_core_utils.health_check_helpers import HealthCheckHelpers, default_health_check_mode
from litellm.litellm_core_utils.health_check_utils import OPTIONAL_STR
# Use cached import helper to lazy-load Logging class (only loads when function is called)
Logging: Final = get_litellm_logging_class()
@ -8765,28 +8766,25 @@ async def ahealth_check(
)
#########################################################
try:
model: str | None = model_params.get("model", None)
if model is None:
requested_model: Final = OPTIONAL_STR.validate_python(model_params.get("model", None))
if requested_model is None:
raise Exception("model not set")
if model in litellm.model_cost and mode is None:
mode = litellm.model_cost[model].get("mode")
custom_llm_provider_from_params: Final = model_params.get("custom_llm_provider", None)
api_base_from_params: Final = model_params.get("api_base", None)
api_key_from_params: Final = model_params.get("api_key", None)
model, custom_llm_provider, _, _ = get_llm_provider(
model=model,
model=requested_model,
custom_llm_provider=custom_llm_provider_from_params,
api_base=api_base_from_params,
api_key=api_key_from_params,
)
if model in litellm.model_cost and mode is None:
mode = litellm.model_cost[model].get("mode")
model_params["cache"] = {"no-cache": True} # don't used cached responses for making health check calls
mode = mode or "chat"
mode = mode or default_health_check_mode(
requested_model=requested_model, model=model, custom_llm_provider=custom_llm_provider
)
if "*" in model:
return await HealthCheckHelpers.ahealth_check_wildcard_models(
model=model,
@ -8815,12 +8813,6 @@ async def ahealth_check(
if isinstance(stack_trace, str):
stack_trace = stack_trace[:1000]
if mode is None:
return {
"error": f"error:{e}. Missing `mode`. Set the `mode` for the model - https://docs.litellm.ai/docs/proxy/health#embedding-models \nstacktrace: {stack_trace}",
"exception": e,
}
error_to_return: Final = str(e) + "\nstack trace: " + stack_trace
raw_request_typed_dict: Final = litellm_logging_obj.model_call_details.get("raw_request_typed_dict")

View file

@ -26,6 +26,7 @@ from litellm.constants import (
DEFAULT_HEALTH_CHECK_PROMPT,
HEALTH_CHECK_TIMEOUT_SECONDS,
)
from litellm.litellm_core_utils.health_check_helpers import native_health_check_mode
from litellm.router_utils.auto_router_model_naming import (
StrategyRouterDependency,
classify_strategy_router_model,
@ -69,18 +70,29 @@ HEALTH_DISPLAY_PARAMS: Final = (
# endpoints that reject unknown fields with 400 "Unknown parameter:
# 'max_tokens'". Allow-list so new modes are safe by default.
# Per-deployment override: `model_info.health_check_supports_max_tokens`.
_MAX_TOKEN_SUPPORT_MODES: Final[frozenset[str]] = frozenset({"chat", "completion", "responses"})
_MAX_TOKEN_SUPPORT_MODES: Final[frozenset[str]] = frozenset({"chat", "completion", "responses", "anthropic_messages"})
def _resolve_health_check_mode(model_info: Mapping[str, object], litellm_params: Mapping[str, object]) -> str | None:
def _native_health_check_mode(model: str, provider_param: object) -> str | None:
try:
resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider(
model=model, custom_llm_provider=provider_param if isinstance(provider_param, str) else None
)
except Exception:
return None
return native_health_check_mode(model=resolved_model, custom_llm_provider=custom_llm_provider)
def resolve_health_check_mode(model_info: Mapping[str, object], litellm_params: Mapping[str, object]) -> str | None:
"""
Effective mode for a deployment's health-check probe.
Prefers operator-set `model_info.mode`; otherwise resolves it from the model
cost map, which understands `bedrock/` and cross-region inference-profile
prefixes (`us.`, `eu.`, `apac.`). Without this, non-chat Bedrock deployments
(e.g. embeddings) are probed as chat, so `max_tokens` is injected and the
request 400s on "extraneous key [max_tokens]".
Prefers operator-set `model_info.mode`; then the mode the provider requires for
that model family (Bedrock Mantle serves Claude ids on the Messages API only);
otherwise resolves it from the model cost map, which understands `bedrock/` and
cross-region inference-profile prefixes (`us.`, `eu.`, `apac.`). Without this,
non-chat Bedrock deployments (e.g. embeddings) are probed as chat, so
`max_tokens` is injected and the request 400s on "extraneous key [max_tokens]".
"""
explicit_mode: Final = model_info.get("mode")
if isinstance(explicit_mode, str):
@ -88,6 +100,9 @@ def _resolve_health_check_mode(model_info: Mapping[str, object], litellm_params:
model: Final = litellm_params.get("model")
if not isinstance(model, str):
return None
native_mode: Final = _native_health_check_mode(model, litellm_params.get("custom_llm_provider"))
if native_mode is not None:
return native_mode
try:
return litellm.get_model_info(model=model).get("mode")
except Exception:
@ -518,7 +533,7 @@ async def _run_model_health_check(model: dict):
if _is_strategy_router_deployment(litellm_params):
return {}
mode: Final = _resolve_health_check_mode(
mode: Final = resolve_health_check_mode(
model_info,
litellm_params, # any-ok: untyped router config dict
)
@ -768,7 +783,7 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
- updates the `voice` param with the `health_check_voice` for `audio_speech` mode if it exists Doc: https://docs.litellm.ai/docs/proxy/health#text-to-speech-models
- for Bedrock models with region routing (bedrock/region/model), strips the litellm routing prefix but preserves the model ID, and pins `custom_llm_provider` to `bedrock` (only when the deployment hasn't already set one, so an explicit `bedrock_converse` survives) so the bare model id still resolves to the provider (e.g. cross-region ids like `us.cohere.embed-v4:0`)
"""
mode: Final = _resolve_health_check_mode(
mode: Final = resolve_health_check_mode(
model_info,
litellm_params, # any-ok: untyped router config dict
)

View file

@ -12,6 +12,7 @@ from typing import Any, Final, Literal, TypedDict, cast
import fastapi
from fastapi import APIRouter, Depends, HTTPException, Request, Response, status
from pydantic import TypeAdapter
from typing_extensions import ReadOnly
import litellm
@ -58,6 +59,7 @@ from litellm.proxy.health_check import (
deployments_targeted_by_name,
health_check_filter_kwargs_from_general_settings,
perform_health_check,
resolve_health_check_mode,
run_with_timeout,
)
from litellm.proxy.middleware.admission_control_middleware import (
@ -173,6 +175,24 @@ def _config_base_for_health_check(
return {key: value for key, value in config_params.items() if key not in _CONFIG_CONNECTION_FIELDS}
def _model_info_for_mode_resolution(
model_info: Mapping[str, object], stored_params: Mapping[str, object], request_params: Mapping[str, object]
) -> Mapping[str, object]:
stored_model: Final = stored_params.get("model")
if stored_model is None or request_params.get("model") in (None, stored_model):
return model_info
return {key: value for key, value in model_info.items() if key != "mode"}
def _string_mode_or_bad_request(params_mode: object) -> str | None:
if params_mode is None or isinstance(params_mode, str):
return params_mode
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={"error": f"litellm_params.mode must be a string, got {type(params_mode).__name__}"},
)
def get_callback_identifier(callback):
"""
Get the callback identifier string, handling both strings and objects.
@ -203,6 +223,7 @@ def get_callback_identifier(callback):
router: Final = APIRouter()
_OBJECT_MAPPING: Final = TypeAdapter(Mapping[str, object])
services = (
Literal[
"slack_budget_alerts",
@ -2033,11 +2054,16 @@ async def test_model_connection(
"rerank",
"realtime",
"responses",
"anthropic_messages",
"ocr",
]
| None = fastapi.Body(
None,
description="The mode to test the model with. If not provided, auto-detected from model capabilities.",
description=(
"The mode to test the model with. If not provided, resolved the way /health does: the deployment's "
"model_info.mode (only while the request tests the deployment's own model), then the mode the "
"provider requires for that model, then the model cost map."
),
),
litellm_params: dict = fastapi.Body(
None,
@ -2188,8 +2214,13 @@ async def test_model_connection(
}
resolved_model_info: Final = loaded_model_info if loaded_model_info is not None else model_info
probe_model_info: Final = _model_info_for_mode_resolution(
_OBJECT_MAPPING.validate_python(resolved_model_info or {}),
stored_params=_OBJECT_MAPPING.validate_python(config_litellm_params),
request_params=_OBJECT_MAPPING.validate_python(request_litellm_params),
)
litellm_params = _update_litellm_params_for_health_check(
model_info=resolved_model_info or {},
model_info=dict(probe_model_info),
litellm_params=litellm_params,
)
@ -2204,12 +2235,17 @@ async def test_model_connection(
prisma_client=prisma_client,
premium_user=premium_user,
)
mode = mode or litellm_params.pop("mode", None)
raw_params_mode: Final[object] = litellm_params.pop("mode", None)
probe_mode: Final = (
mode
or _string_mode_or_bad_request(raw_params_mode)
or resolve_health_check_mode(probe_model_info, _OBJECT_MAPPING.validate_python(litellm_params))
)
result: Final = await run_with_timeout(
litellm.ahealth_check(
model_params=litellm_params,
mode=mode,
mode=probe_mode,
prompt="test from litellm",
input=["test from litellm"],
),

View file

@ -16,6 +16,8 @@ from litellm.constants import LITTELM_INTERNAL_HEALTH_SERVICE_ACCOUNT_NAME
from litellm.litellm_core_utils.health_check_helpers import (
IMAGE_EDIT_HEALTH_CHECK_PROMPT,
HealthCheckHelpers,
default_health_check_mode,
native_health_check_mode,
)
from litellm.main import ahealth_check
from litellm.proxy._types import UserAPIKeyAuth
@ -647,3 +649,162 @@ async def test_ahealth_check_probes_strands_through_decisions_without_mode(
assert "error" not in result, result
assert upstream.called
assert "authorization" not in upstream.calls[0].request.headers
@pytest.mark.parametrize(
("model", "custom_llm_provider", "expected"),
(
("anthropic.claude-haiku-4-5", "bedrock_mantle", "anthropic_messages"),
("Anthropic.Claude-Opus-5-5", "bedrock_mantle", "anthropic_messages"),
("openai.gpt-oss-120b", "bedrock_mantle", None),
("us.anthropic.claude-haiku-4-5-20251001-v1:0", "bedrock", None),
("claude-haiku-4-5", "anthropic", None),
("anthropic.claude-haiku-4-5", None, None),
),
)
def test_native_health_check_mode_is_messages_only_for_mantle_claude(
model: str, custom_llm_provider: str | None, expected: str | None
) -> None:
assert native_health_check_mode(model=model, custom_llm_provider=custom_llm_provider) == expected
def test_default_health_check_mode_prefers_the_native_surface_over_the_cost_map(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(litellm, "model_cost", {"anthropic.claude-haiku-4-5": {"mode": "chat"}})
assert (
default_health_check_mode(
requested_model="bedrock_mantle/anthropic.claude-haiku-4-5",
model="anthropic.claude-haiku-4-5",
custom_llm_provider="bedrock_mantle",
)
== "anthropic_messages"
)
@pytest.mark.parametrize(
("model_cost", "expected"),
(
({"bedrock_mantle/openai.gpt-oss-120b": {"mode": "responses"}}, "responses"),
({"openai.gpt-oss-120b": {"mode": "completion"}}, "completion"),
(
{
"bedrock_mantle/openai.gpt-oss-120b": {"mode": "responses"},
"openai.gpt-oss-120b": {"mode": "completion"},
},
"responses",
),
({}, "chat"),
),
)
def test_default_health_check_mode_falls_back_to_cost_map_then_chat(
model_cost: dict[str, dict[str, str]], expected: str, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(litellm, "model_cost", model_cost)
assert (
default_health_check_mode(
requested_model="bedrock_mantle/openai.gpt-oss-120b",
model="openai.gpt-oss-120b",
custom_llm_provider="bedrock_mantle",
)
== expected
)
@pytest.mark.asyncio
@pytest.mark.parametrize("mode_kwargs", [{}, {"mode": None}], ids=["omitted", "explicit_none"])
async def test_ahealth_check_probes_mantle_claude_through_messages_without_mode(
monkeypatch: pytest.MonkeyPatch,
respx_mock: respx.MockRouter,
mode_kwargs: dict[str, None],
) -> None:
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
litellm.in_memory_llm_clients_cache.flush_cache()
upstream: Final = respx_mock.post("https://bedrock-mantle.us-east-2.api.aws/anthropic/v1/messages").respond(
json={
"id": "msg_health",
"type": "message",
"role": "assistant",
"model": "anthropic.claude-haiku-4-5",
"content": [{"type": "text", "text": "pong"}],
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 3, "output_tokens": 1},
}
)
result: Final = await ahealth_check(
{
"model": "bedrock_mantle/anthropic.claude-haiku-4-5",
"api_key": "test-bearer",
"aws_region_name": "us-east-2",
},
prompt="test from litellm",
**mode_kwargs,
)
assert "error" not in result, result
assert upstream.call_count == 1
sent: Final = json.loads(upstream.calls.last.request.content)
assert sent["model"] == "anthropic.claude-haiku-4-5"
assert sent["max_tokens"] == 16
assert sent["messages"] == [{"role": "user", "content": "test from litellm"}]
@pytest.mark.asyncio
async def test_ahealth_check_anthropic_messages_mode_keeps_caller_supplied_messages(
monkeypatch: pytest.MonkeyPatch,
respx_mock: respx.MockRouter,
) -> None:
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
litellm.in_memory_llm_clients_cache.flush_cache()
upstream: Final = respx_mock.post("https://bedrock-mantle.us-east-2.api.aws/anthropic/v1/messages").respond(
json={
"id": "msg_health",
"type": "message",
"role": "assistant",
"model": "anthropic.claude-haiku-4-5",
"content": [{"type": "text", "text": "pong"}],
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 3, "output_tokens": 1},
}
)
result: Final = await ahealth_check(
{
"model": "bedrock_mantle/anthropic.claude-haiku-4-5",
"api_key": "test-bearer",
"aws_region_name": "us-east-2",
"messages": [{"role": "user", "content": "operator probe"}],
"max_tokens": 4,
},
mode="anthropic_messages",
prompt="test from litellm",
)
assert "error" not in result, result
sent: Final = json.loads(upstream.calls.last.request.content)
assert sent["max_tokens"] == 4
assert sent["messages"] == [{"role": "user", "content": "operator probe"}]
@pytest.mark.asyncio
@pytest.mark.parametrize(
("model_params", "expected_error"),
(
({"model": "not-a-provider/some-model"}, "LLM Provider NOT provided"),
({"api_key": "test-bearer"}, "model not set"),
),
ids=["unknown_provider", "model_missing"],
)
async def test_ahealth_check_without_mode_reports_the_real_failure(
model_params: dict[str, str], expected_error: str
) -> None:
result: Final = await ahealth_check(model_params, prompt="test from litellm")
assert expected_error in result["error"], result["error"]
assert "Missing `mode`" not in result["error"]
assert "raw_request_typed_dict" in result

View file

@ -5,14 +5,14 @@ import time
from collections.abc import Iterator, Mapping, Sequence
from contextlib import contextmanager
from datetime import datetime, timedelta
from types import SimpleNamespace
from types import MappingProxyType, SimpleNamespace
from typing import Final
from unittest.mock import AsyncMock, MagicMock, patch
import httpx
import pytest
import respx
from fastapi import FastAPI
from fastapi import FastAPI, HTTPException
from fastapi.testclient import TestClient
from prisma.errors import ClientNotConnectedError, HTTPClientClosedError, PrismaError
@ -694,6 +694,181 @@ async def test_test_model_connection_falls_back_to_deployments_zero_without_id()
assert model_params.get("api_key") == "fake-key-A"
@contextmanager
def _test_connection_probe(
deployment: Mapping[str, object],
) -> Iterator[AsyncMock]:
from litellm.types.router import Deployment, LiteLLM_Params
router: Final = MagicMock()
router.get_deployment.side_effect = lambda model_id: (
Deployment(
model_name=str(deployment["model_name"]),
litellm_params=LiteLLM_Params(**deployment["litellm_params"]), # pyright: ignore[reportArgumentType] # test fixture dict
model_info=deployment["model_info"], # pyright: ignore[reportArgumentType] # test fixture dict
)
if model_id == deployment["model_info"]["id"] # pyright: ignore[reportIndexIssue] # test fixture dict
else None
)
ahealth_check: Final = AsyncMock(return_value={"status": "healthy"})
with (
patch("litellm.proxy.proxy_server.prisma_client", MagicMock()),
patch("litellm.proxy.proxy_server.llm_router", router),
patch("litellm.proxy.proxy_server.premium_user", False),
patch(
"litellm.proxy.management_endpoints.model_management_endpoints.ModelManagementAuthChecks.can_user_make_model_call",
AsyncMock(),
),
patch("litellm.proxy.health_endpoints._health_endpoints.litellm.ahealth_check", ahealth_check),
patch(
"litellm.proxy.health_endpoints._health_endpoints.run_with_timeout",
AsyncMock(return_value={"status": "healthy"}),
),
):
yield ahealth_check
MANTLE_CLAUDE_DEPLOYMENT: Final = MappingProxyType(
{
"model_name": "claude-haiku-4-5",
"litellm_params": {
"model": "bedrock_mantle/anthropic.claude-haiku-4-5",
"api_key": "fake-mantle-key",
"aws_region_name": "us-east-2",
},
"model_info": {"id": "mantle-claude-id"},
}
)
@pytest.mark.asyncio
async def test_test_model_connection_without_mode_probes_mantle_claude_over_messages():
"""
The Admin UI model page sends the row's id and no mode. The probe must then resolve
the mode the way /health does, so a Bedrock Mantle Claude deployment is checked over
the Anthropic Messages API instead of chat completions, which Mantle rejects.
"""
with _test_connection_probe(MANTLE_CLAUDE_DEPLOYMENT) as ahealth_check:
result: Final = await health_test_model_connection(
request=MagicMock(),
mode=None,
litellm_params={"model": "bedrock_mantle/anthropic.claude-haiku-4-5"},
model_info={"id": "mantle-claude-id"},
user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"),
)
assert result["status"] == "success"
assert ahealth_check.call_args.kwargs["mode"] == "anthropic_messages"
@pytest.mark.asyncio
@pytest.mark.parametrize(
("request_params", "expected_mode"),
[
({"model": "bedrock_mantle/anthropic.claude-haiku-4-5"}, "chat"),
({}, "chat"),
({"model": "bedrock_mantle/anthropic.claude-sonnet-4-5"}, "anthropic_messages"),
],
ids=["stored_model", "no_model", "overridden_model"],
)
async def test_test_model_connection_stored_operator_mode_follows_the_stored_model(
request_params: Mapping[str, str], expected_mode: str
):
"""
A mode the operator stored on the deployment is the probe's mode when the request
carries none, ahead of the provider-native rule, but only while the request probes
the deployment's own model. A request that selects the deployment by id and swaps in
another model resolves the mode from that model instead.
"""
deployment: Final = MappingProxyType(
{**MANTLE_CLAUDE_DEPLOYMENT, "model_info": {"id": "mantle-claude-id", "mode": "chat"}}
)
with _test_connection_probe(deployment) as ahealth_check:
await health_test_model_connection(
request=MagicMock(),
mode=None,
litellm_params=dict(request_params),
model_info={"id": "mantle-claude-id"},
user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"),
)
assert ahealth_check.call_args.kwargs["mode"] == expected_mode
@pytest.mark.asyncio
async def test_test_model_connection_overridden_model_probe_params_follow_the_probed_model():
"""
When the request selects a deployment by id and swaps in another model, the probe's
params are shaped for that model, so the stored mode must not inject `max_tokens`
into what is now an embedding probe (Mistral rejects it with a 422 extra_forbidden).
"""
deployment: Final = MappingProxyType(
{
"model_name": "anthropic-claude-haiku-4-5",
"litellm_params": {"model": "anthropic/claude-haiku-4-5", "api_key": "fake-anthropic-key"},
"model_info": {"id": "anthropic-messages-id", "mode": "anthropic_messages"},
}
)
with _test_connection_probe(deployment) as ahealth_check:
await health_test_model_connection(
request=MagicMock(),
mode=None,
litellm_params={"model": "mistral/mistral-embed", "api_key": "fake-mistral-key"},
model_info={"id": "anthropic-messages-id"},
user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"),
)
assert ahealth_check.call_args.kwargs["mode"] == "embedding"
assert "max_tokens" not in ahealth_check.call_args.kwargs["model_params"]
@pytest.mark.asyncio
@pytest.mark.parametrize("params_mode", [123, ["chat"], {"mode": "chat"}, False], ids=["int", "list", "dict", "bool"])
async def test_test_model_connection_non_string_params_mode_is_a_bad_request(params_mode: object):
with _test_connection_probe(MANTLE_CLAUDE_DEPLOYMENT) as ahealth_check:
with pytest.raises(HTTPException) as exc_info:
await health_test_model_connection(
request=MagicMock(),
mode=None,
litellm_params={"model": "bedrock_mantle/anthropic.claude-haiku-4-5", "mode": params_mode},
model_info={"id": "mantle-claude-id"},
user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"),
)
assert exc_info.value.status_code == 400
assert "litellm_params.mode must be a string" in exc_info.value.detail["error"]
ahealth_check.assert_not_called()
@pytest.mark.asyncio
async def test_test_model_connection_string_params_mode_is_the_probe_mode():
with _test_connection_probe(MANTLE_CLAUDE_DEPLOYMENT) as ahealth_check:
await health_test_model_connection(
request=MagicMock(),
mode=None,
litellm_params={"model": "bedrock_mantle/anthropic.claude-haiku-4-5", "mode": "chat"},
model_info={"id": "mantle-claude-id"},
user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"),
)
assert ahealth_check.call_args.kwargs["mode"] == "chat"
assert "mode" not in ahealth_check.call_args.kwargs["model_params"]
@pytest.mark.asyncio
async def test_test_model_connection_request_mode_wins_over_resolved_mode():
with _test_connection_probe(MANTLE_CLAUDE_DEPLOYMENT) as ahealth_check:
await health_test_model_connection(
request=MagicMock(),
mode="chat",
litellm_params={"model": "bedrock_mantle/anthropic.claude-haiku-4-5"},
model_info={"id": "mantle-claude-id"},
user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"),
)
assert ahealth_check.call_args.kwargs["mode"] == "chat"
@pytest.mark.asyncio
async def test_test_model_connection_uses_loaded_deployment_team_id():
"""

View file

@ -11,7 +11,7 @@ from litellm.proxy import health_check as hc_module
from litellm.proxy.health_check import (
_is_strategy_router_deployment,
_resolve_health_check_max_tokens,
_resolve_health_check_mode,
resolve_health_check_mode,
_update_litellm_params_for_health_check,
)
@ -406,7 +406,7 @@ def test_update_litellm_params_health_check_reasoning_effort():
)
def test_bedrock_embedding_without_explicit_mode_skips_max_tokens(deployment_model, expected_request_model):
"""Embedding mode auto-detected from model cost map -> no max_tokens, provider pinned."""
assert _resolve_health_check_mode({}, {"model": deployment_model}) == "embedding"
assert resolve_health_check_mode({}, {"model": deployment_model}) == "embedding"
updated = _update_litellm_params_for_health_check({}, {"model": deployment_model})
@ -417,12 +417,12 @@ def test_bedrock_embedding_without_explicit_mode_skips_max_tokens(deployment_mod
def test_resolve_health_check_mode_prefers_explicit_model_info_mode():
"""An operator-set mode wins over model-cost lookup."""
assert _resolve_health_check_mode({"mode": "chat"}, {"model": "bedrock/amazon.titan-embed-text-v2:0"}) == "chat"
assert resolve_health_check_mode({"mode": "chat"}, {"model": "bedrock/amazon.titan-embed-text-v2:0"}) == "chat"
def test_resolve_health_check_mode_unknown_model_returns_none():
assert _resolve_health_check_mode({}, {"model": "bedrock/not-a-real-model-xyz"}) is None
assert _resolve_health_check_mode({}, {}) is None
assert resolve_health_check_mode({}, {"model": "bedrock/not-a-real-model-xyz"}) is None
assert resolve_health_check_mode({}, {}) is None
def test_bedrock_chat_without_mode_still_injects_max_tokens_and_pins_provider():
@ -481,6 +481,113 @@ async def test_run_model_health_check_threads_resolved_mode_to_ahealth_check():
assert probed_params["model"] == "amazon.titan-embed-text-v2:0"
_MANTLE_CLAUDE_DEPLOYMENT_PARAMS = {
"model": "bedrock_mantle/anthropic.claude-haiku-4-5",
"api_key": "test-bearer",
"aws_region_name": "us-east-2",
}
def _mantle_anthropic_response() -> dict[str, object]:
return {
"id": "msg_health",
"type": "message",
"role": "assistant",
"model": "anthropic.claude-haiku-4-5",
"content": [{"type": "text", "text": "pong"}],
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 3, "output_tokens": 1},
}
@pytest.mark.parametrize(
"deployment_model",
["bedrock_mantle/anthropic.claude-haiku-4-5", "bedrock_mantle/anthropic.claude-opus-5-5"],
)
def test_mantle_claude_without_mode_resolves_to_anthropic_messages(deployment_model):
"""Mantle only serves Claude over /anthropic/v1/messages, so that is the probe surface by default."""
assert resolve_health_check_mode({}, {"model": deployment_model}) == "anthropic_messages"
updated = _update_litellm_params_for_health_check({}, {"model": deployment_model})
assert updated["max_tokens"] == 16
assert [message["role"] for message in updated["messages"]] == ["user"]
def test_mantle_claude_with_explicit_provider_param_resolves_to_anthropic_messages():
assert (
resolve_health_check_mode({}, {"model": "anthropic.claude-haiku-4-5", "custom_llm_provider": "bedrock_mantle"})
== "anthropic_messages"
)
def test_mantle_claude_explicit_chat_mode_wins_over_the_native_default():
assert resolve_health_check_mode({"mode": "chat"}, {"model": "bedrock_mantle/anthropic.claude-haiku-4-5"}) == "chat"
@pytest.mark.parametrize(
"deployment_model",
[
"bedrock_mantle/openai.gpt-oss-120b",
"bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
"anthropic/claude-haiku-4-5",
],
)
def test_native_messages_default_is_scoped_to_mantle_claude(deployment_model):
"""Non-Claude Mantle ids and Claude on other providers keep their chat-completions probe."""
assert resolve_health_check_mode({}, {"model": deployment_model}) == "chat"
@pytest.mark.asyncio
async def test_run_model_health_check_probes_mantle_claude_over_messages(monkeypatch):
"""The deployment the ticket describes, probed end to end through the proxy's health runner.
Before the fix the probe went to /v1/chat/completions, which Mantle answers with a
validation_error for Claude ids, so every such deployment showed unhealthy.
"""
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
litellm.in_memory_llm_clients_cache.flush_cache()
with respx.mock(assert_all_called=False) as respx_mock:
messages_route = respx_mock.post("https://bedrock-mantle.us-east-2.api.aws/anthropic/v1/messages").respond(
json=_mantle_anthropic_response()
)
chat_route = respx_mock.post("https://bedrock-mantle.us-east-2.api.aws/v1/chat/completions").respond(
status_code=400, json={"type": "error", "error": {"type": "validation_error"}}
)
result = await hc_module._run_model_health_check(
{"litellm_params": dict(_MANTLE_CLAUDE_DEPLOYMENT_PARAMS), "model_info": {}}
)
assert "error" not in result, result
assert chat_route.call_count == 0
assert messages_route.call_count == 1
sent = messages_route.calls.last.request
assert sent.headers["authorization"] == "Bearer test-bearer"
body = json.loads(sent.content)
assert body["model"] == "anthropic.claude-haiku-4-5"
assert body["max_tokens"] == 16
assert [message["role"] for message in body["messages"]] == ["user"]
@pytest.mark.asyncio
async def test_run_model_health_check_honors_an_explicit_chat_mode_on_mantle_claude(monkeypatch):
"""Negative control: an operator who pins mode=chat still gets the chat completions probe.
Since #43646 Mantle serves Claude chat completions over its Messages endpoint as well, so
the wire no longer tells the two probes apart and the probe mode is read off the health call.
"""
fake_ahealth_check = AsyncMock(return_value={})
monkeypatch.setattr(litellm, "ahealth_check", fake_ahealth_check)
await hc_module._run_model_health_check(
{"litellm_params": dict(_MANTLE_CLAUDE_DEPLOYMENT_PARAMS), "model_info": {"mode": "chat"}}
)
assert fake_ahealth_check.call_args.kwargs["mode"] == "chat"
def test_autodetected_embedding_skips_reasoning_effort():
"""reasoning_effort must not leak into an embedding probe whose mode is auto-detected.

View file

@ -11,6 +11,7 @@ export const TEST_MODES = [
{ value: "rerank", label: "Rerank - /rerank" },
{ value: "realtime", label: "Realtime - /realtime" },
{ value: "batch", label: "Batch - /batch" },
{ value: "anthropic_messages", label: "Anthropic Messages - /v1/messages" },
{ value: "ocr", label: "OCR - /ocr" },
];

View file

@ -295,6 +295,35 @@ describe("ModelInfoView", () => {
expect(modelInfoArg.id).toBe("123");
});
it("does not echo the displayed mode into the test connection request", async () => {
// /model/info fills model_info.mode in from the cost map for display. Sending that
// value back would pin the probe to it and skip the mode the provider requires,
// so the page forwards only the row's id and lets the proxy resolve the mode.
const user = userEvent.setup();
const displayedModel = {
...defaultModelData,
litellm_params: { ...defaultModelData.litellm_params, model: "bedrock_mantle/anthropic.claude-haiku-4-5" },
model_info: { ...defaultModelData.model_info, mode: "chat", key: "anthropic.claude-haiku-4-5" },
};
mockUseModelsInfo.mockReturnValue({ data: { data: [displayedModel] }, isLoading: false, error: null });
mockModelInfoV1Call.mockResolvedValue({ data: [displayedModel] });
render(<ModelInfoView {...DEFAULT_ADMIN_PROPS} />, { wrapper });
await waitFor(() => {
expect(screen.getByText("Model Settings")).toBeInTheDocument();
});
await user.click(screen.getByRole("button", { name: /test connection/i }));
await waitFor(() => {
expect(mockTestConnectionRequest).toHaveBeenCalled();
});
const [, , modelInfoArg, modeArg] = mockTestConnectionRequest.mock.calls[0];
expect(modelInfoArg).toEqual({ id: "123" });
expect(modeArg).toBeUndefined();
});
it("should display error notification when connection test fails", async () => {
const user = userEvent.setup();
mockTestConnectionRequest.mockRejectedValue(new Error("Connection failed"));

View file

@ -491,9 +491,7 @@ export default function ModelInfoView({
// backend silently falls back to deployments[0] and probes
// the wrong endpoint.
id: localModelData.model_info?.id,
mode: localModelData.model_info?.mode,
},
localModelData.model_info?.mode,
);
if (response.status === "success") {

View file

@ -2165,7 +2165,7 @@ export const testConnectionRequest = async (
accessToken: string,
litellm_params: Record<string, any>,
model_info: Record<string, any>,
mode: string,
mode?: string,
) => {
try {
// Construct the URL based on environment

View file

@ -27407,9 +27407,9 @@ export interface components {
};
/**
* Mode
* @description The mode to test the model with. If not provided, auto-detected from model capabilities.
* @description The mode to test the model with. If not provided, resolved the way /health does: the deployment's model_info.mode (only while the request tests the deployment's own model), then the mode the provider requires for that model, then the model cost map.
*/
mode?: ("chat" | "completion" | "embedding" | "audio_speech" | "audio_transcription" | "image_generation" | "image_edit" | "video_generation" | "batch" | "rerank" | "realtime" | "responses" | "ocr") | null;
mode?: ("chat" | "completion" | "embedding" | "audio_speech" | "audio_transcription" | "image_generation" | "image_edit" | "video_generation" | "batch" | "rerank" | "realtime" | "responses" | "anthropic_messages" | "ocr") | null;
/**
* Model Info
* @description Model info for the health check