fix: increase health check max_tokens from 5 to 16 (#23836) (#26610)

GPT-5 models enforce a minimum of 16 for max_output_tokens. The current
default of 5 still causes health checks to fail for these models. Bump
the non-wildcard default to 16 — the smallest value that satisfies all
known provider minimums while keeping health checks lightweight.

Also tightens the wildcard test assertion from a weak disjunctive check
to strict key-absence.

Co-authored-by: Sameer Kankute <sameer@berri.ai>
This commit is contained in:
Hannah Smith 2026-06-18 08:06:18 -04:00 • committed by GitHub
parent 08166b83f9
commit 18564992a3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 3 additions and 4 deletions

View file

@ -401,7 +401,7 @@ def _resolve_health_check_max_tokens(
3. For non-wildcard reasoning routes: BACKGROUND_HEALTH_CHECK_MAX_TOKENS_REASONING
from env (if set)
4. BACKGROUND_HEALTH_CHECK_MAX_TOKENS (global, any route including wildcards)
5. Non-wildcard default: 5
5. Non-wildcard default: 16
6. Wildcard and nothing from (1)(4): leave unset (caller omits max_tokens)
"""
explicit = model_info.get("health_check_max_tokens", None)
@ -432,7 +432,7 @@ def _resolve_health_check_max_tokens(
return int(BACKGROUND_HEALTH_CHECK_MAX_TOKENS)
if not is_wildcard:
return 16 # OpenAI GPT-5 models require max_output_tokens >= 16
return 16
return None

View file

@ -49,8 +49,7 @@ async def test_update_litellm_params_max_tokens_wildcard():
updated_params = _update_litellm_params_for_health_check(model_info, litellm_params)
# Should not be set to 1
assert "max_tokens" not in updated_params or updated_params["max_tokens"] != 1
assert "max_tokens" not in updated_params
@pytest.mark.asyncio