fix(health): drop model capability flags from health-check probes

supports_* and related model_info metadata were ending up on the probe
kwargs and Bedrock rejected them as Extra inputs. Copy params before
mutating and strip those keys before the call.

Fixes #38941
This commit is contained in:
lei_lei 2026-08-31 08:47:43 +00:00 • committed by lei_lei
parent c2c2a623c0
commit f5b159e38d
2 changed files with 69 additions and 1 deletions

View file

@ -112,6 +112,30 @@ def _should_inject_health_check_max_tokens(model_info: Mapping[str, object], mod
# Health-check modes that forward `reasoning_effort` to the provider (chat-style calls).
_HEALTH_CHECK_MODES_SUPPORTING_REASONING_EFFORT: Final = frozenset((None, "chat", "completion"))
# Model-map / model_info capability metadata that must never ride on the health-check
# probe as provider request fields. Bedrock (and similar) reject them with
# ``supports_*: Extra inputs are not permitted`` (#38941). They reach the body via
# ``add_provider_specific_params_to_optional_params`` → ``{**optional_params}``.
_HEALTH_CHECK_MODEL_METADATA_KEYS: Final = frozenset(
{
"reasoning_effort_levels",
"default_reasoning_effort",
"bedrock_output_config_effort_ceiling",
"bedrock_converse_supports_strict_tools",
"thinking_always_on",
}
)
def _is_health_check_model_metadata_key(key: str) -> bool:
"""True for model capability / catalog metadata that is not a request param."""
return key.startswith("supports_") or key in _HEALTH_CHECK_MODEL_METADATA_KEYS
def _strip_model_metadata_from_health_params(litellm_params: dict) -> dict:
"""Drop model_info capability keys that leaked onto the health-check probe params."""
return {k: v for k, v in litellm_params.items() if not _is_health_check_model_metadata_key(k)}
def _get_process_rss_mb() -> float | None:
"""
@ -757,6 +781,7 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
"""
Update the litellm params for health check.
- copies `litellm_params` so the shared deployment dict is not mutated
- merges `model_info.health_check_params` into the probe request, so a deployment whose provider
requires a payload field litellm does not synthesize (e.g. `mediaSource` for Bedrock TwelveLabs
Pegasus) can supply it. The dedicated knobs below are applied afterwards and win on conflict.
@ -769,7 +794,11 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
- updates the `model` param with the `health_check_model` if it exists Doc: https://docs.litellm.ai/docs/proxy/health#wildcard-routes
- updates the `voice` param with the `health_check_voice` for `audio_speech` mode if it exists Doc: https://docs.litellm.ai/docs/proxy/health#text-to-speech-models
- for Bedrock models with region routing (bedrock/region/model), strips the litellm routing prefix but preserves the model ID, and pins `custom_llm_provider` to `bedrock` (only when the deployment hasn't already set one, so an explicit `bedrock_converse` survives) so the bare model id still resolves to the provider (e.g. cross-region ids like `us.cohere.embed-v4:0`)
- strips model capability metadata (`supports_*`, effort ceilings, …) so it cannot leak into
the provider request body (#38941)
"""
# Copy first: callers pass the live deployment litellm_params dict.
litellm_params = dict(litellm_params)
mode: Final = _resolve_health_check_mode(
model_info,
litellm_params, # any-ok: untyped router config dict
@ -843,7 +872,7 @@ def _update_litellm_params_for_health_check(model_info: dict, litellm_params: di
"bedrock"
)
return litellm_params
return _strip_model_metadata_from_health_params(litellm_params)
async def perform_health_check(

View file

@ -832,3 +832,42 @@ async def test_health_check_with_custom_llm_provider():
# Should succeed without "LLM Provider NOT provided" error
assert "error" not in response
assert isinstance(response, dict)
def test_health_check_strips_model_capability_metadata():
"""
Capability flags on litellm_params / health_check_params must not ride the probe.
Bedrock rejects them as request fields (#38941):
``supports_max_reasoning_effort: Extra inputs are not permitted``.
Also ensure the shared deployment dict is not mutated.
"""
from litellm.proxy.health_check import _update_litellm_params_for_health_check
original = {
"model": "bedrock/anthropic.claude-3-7-sonnet-20240620-v1:0",
"api_key": "fake_key",
"supports_max_reasoning_effort": True,
"supports_xhigh_reasoning_effort": None,
"bedrock_output_config_effort_ceiling": "xhigh",
"thinking": {"type": "enabled", "budget_tokens": 1024},
}
model_info = {
"health_check_params": {
"supports_max_reasoning_effort": True,
"mediaSource": {"s3Location": {"uri": "s3://bucket/key"}},
},
}
updated = _update_litellm_params_for_health_check(model_info, original)
assert "supports_max_reasoning_effort" not in updated
assert "supports_xhigh_reasoning_effort" not in updated
assert "bedrock_output_config_effort_ceiling" not in updated
# Legitimate probe / provider fields stay
assert "messages" in updated
assert updated["thinking"] == {"type": "enabled", "budget_tokens": 1024}
assert updated["mediaSource"] == {"s3Location": {"uri": "s3://bucket/key"}}
# Shared deployment dict must stay untouched
assert "messages" not in original
assert original["supports_max_reasoning_effort"] is True