From 68c891cf2f7d7d4dc0eb65d2eb8b8ebe8588c371 Mon Sep 17 00:00:00 2001 From: Rob Sherman Date: Sat, 28 Feb 2026 20:20:34 -0800 Subject: [PATCH] fix(anthropic-adapter): cap max_tokens against model output limit in messages adapter When routing non-Anthropic models (e.g. Amazon Bedrock Nova Pro) through the Anthropic-compatible /v1/messages endpoint, `_prepare_completion_kwargs` was forwarding the client-supplied max_tokens directly to the underlying provider without checking the model's actual output token limit. This caused hard failures for models like Amazon Nova Pro (10,000 token limit) when clients such as Claude Code send large max_tokens values that are valid for other providers (e.g. 64,000 for Anthropic Claude). Fix: look up max_output_tokens via litellm.get_model_info() before building the request_data dict and cap accordingly. The lookup uses custom_llm_provider from extra_kwargs (already resolved by the outer anthropic_messages_handler via get_llm_provider) with a fallback to inferring the provider from the model string. Tested with bedrock/converse/us.amazon.nova-pro-v1:0 routed via LiteLLM proxy: requests sending max_tokens=16000 are now silently capped to 10000 and succeed instead of returning HTTP 400. --- .../adapters/handler.py | 35 +++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index 73e74c228ba..c1629e01841 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -121,6 +121,41 @@ class LiteLLMMessagesToCompletionTransformationHandler: Logging as LiteLLMLoggingObject, ) + # Cap max_tokens against the model's known output token limit. + # Without this, requests from clients like Claude Code (which may send + # large max_tokens values valid for other providers) will be rejected by + # models with stricter limits (e.g. Amazon Nova Pro: 10,000 tokens). + # + # By this point get_llm_provider() has already been called in the outer + # anthropic_messages_handler, so `model` is the stripped model name + # (e.g. "converse/us.amazon.nova-pro-v1:0") and `custom_llm_provider` + # is passed via extra_kwargs (e.g. "bedrock"). We try the lookup with + # the provider first, then fall back to inferring it from the model + # string for cases where extra_kwargs doesn't carry the provider. + _custom_llm_provider = (extra_kwargs or {}).get("custom_llm_provider") + _capped = False + try: + model_info = litellm.get_model_info( + model=model, custom_llm_provider=_custom_llm_provider + ) + model_max_output = model_info.get("max_output_tokens") + if model_max_output and max_tokens > model_max_output: + max_tokens = model_max_output + _capped = True + except Exception: + pass + if not _capped: + try: + _, inferred_provider, _, _ = litellm.utils.get_llm_provider(model) + model_info = litellm.get_model_info( + model=model, custom_llm_provider=inferred_provider + ) + model_max_output = model_info.get("max_output_tokens") + if model_max_output and max_tokens > model_max_output: + max_tokens = model_max_output + except Exception: + pass + request_data = { "model": model, "messages": messages,