diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 39d612f252d..de6c8f33f75 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -1652,7 +1652,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1681,7 +1681,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1710,7 +1710,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1739,7 +1739,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1768,7 +1768,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1797,7 +1797,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2470,7 +2470,7 @@ "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -10240,7 +10240,7 @@ "search_context_size_medium": 0.01 }, "supports_adaptive_thinking": true, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -34942,7 +34942,7 @@ "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -42350,7 +42350,7 @@ "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, diff --git a/litellm/router.py b/litellm/router.py index e54eadfb872..26ef92edc57 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -108,6 +108,7 @@ from litellm.router_utils.cooldown_handlers import ( ) from litellm.router_utils.fallback_event_handlers import ( _check_non_standard_fallback_format, + build_mid_stream_continuation_messages, get_fallback_model_group, run_async_fallback, ) @@ -2233,17 +2234,16 @@ class Router: # would waste tokens and confuse the model. initial_kwargs["messages"] = messages else: - initial_kwargs["messages"] = messages + [ - { - "role": "system", - "content": "You are a helpful assistant. You are given a message and you need to respond to it. You are also given a generated content. You need to respond to the message in continuation of the generated content. Do not repeat the same content. Your response should be in continuation of this text: ", - }, - { - "role": "assistant", - "content": e.generated_content, - "prefix": True, - }, - ] + # Prefill-resume only where the model accepts assistant + # prefill (Claude Sonnet 4.6+/Opus 4.6+ return a 400 for + # it); otherwise the partial text rides a user message. + initial_kwargs["messages"] = ( + build_mid_stream_continuation_messages( + messages=messages, + generated_content=e.generated_content, + model_group=model_group, + ) + ) self._update_kwargs_before_fallbacks( model=model_group, kwargs=initial_kwargs ) @@ -2793,17 +2793,16 @@ class Router: if e.is_pre_first_chunk or not e.generated_content: initial_kwargs["messages"] = messages else: - initial_kwargs["messages"] = messages + [ - { - "role": "system", - "content": "You are a helpful assistant. You are given a message and you need to respond to it. You are also given a generated content. You need to respond to the message in continuation of the generated content. Do not repeat the same content. Your response should be in continuation of this text: ", - }, - { - "role": "assistant", - "content": e.generated_content, - "prefix": True, - }, - ] + # Prefill-resume only where the model accepts assistant + # prefill (Claude Sonnet 4.6+/Opus 4.6+ return a 400 for + # it); otherwise the partial text rides a user message. + initial_kwargs["messages"] = ( + build_mid_stream_continuation_messages( + messages=messages, + generated_content=e.generated_content, + model_group=model_group, + ) + ) router_self._update_kwargs_before_fallbacks( model=model_group, kwargs=initial_kwargs ) diff --git a/litellm/router_utils/fallback_event_handlers.py b/litellm/router_utils/fallback_event_handlers.py index eb756e3cf8b..3bc609a7333 100644 --- a/litellm/router_utils/fallback_event_handlers.py +++ b/litellm/router_utils/fallback_event_handlers.py @@ -17,6 +17,70 @@ else: LitellmRouter = Any +MID_STREAM_CONTINUATION_SYSTEM_PROMPT = "You are a helpful assistant. You are given a message and you need to respond to it. You are also given a generated content. You need to respond to the message in continuation of the generated content. Do not repeat the same content. Your response should be in continuation of this text: " + + +def _prefill_explicitly_unsupported(model_group: Optional[str]) -> bool: + """True only when the model registry explicitly marks the model as NOT + supporting assistant prefill (``supports_assistant_prefill: false``). + + Absent/unknown capability returns False so models without registry data keep + the legacy prefill behavior — only models that would reject the prefill with + a 400 anyway are routed to the user-message continuation. + """ + if model_group is None: + return False + try: + from litellm.utils import get_model_info + + model_info = get_model_info(model=model_group) + return model_info.get("supports_assistant_prefill") is False + except Exception: + return False + + +def build_mid_stream_continuation_messages( + messages: List[Any], + generated_content: str, + model_group: Optional[str], +) -> List[Any]: + """Build the message list a mid-stream fallback uses to resume an interrupted stream. + + By default the partial response is appended as a prefixed assistant message, + so the fallback model continues the text exactly where the stream died. + + Anthropic removed assistant prefill starting with Claude Sonnet 4.6 / Opus 4.6 — + a prefilled assistant message returns a 400 error, which made every mid-stream + fallback for those models fail deterministically. For models the registry + explicitly marks as not supporting prefill, the partial response is quoted in + a trailing USER message instead — the continuation pattern Anthropic's + migration guide documents: + https://platform.claude.com/docs/en/about-claude/models/migration-guide + """ + if _prefill_explicitly_unsupported(model_group): + return messages + [ + { + "role": "user", + "content": ( + "Your previous response was interrupted and ended with:\n" + f"{generated_content}\n" + "Continue from where you left off. Do not repeat the content already generated." + ), + } + ] + return messages + [ + { + "role": "system", + "content": MID_STREAM_CONTINUATION_SYSTEM_PROMPT, + }, + { + "role": "assistant", + "content": generated_content, + "prefix": True, + }, + ] + + def _check_stripped_model_group(model_group: str, fallback_key: str) -> bool: """ Handles wildcard routing scenario diff --git a/litellm/utils.py b/litellm/utils.py index 916260cab5a..a1fb87e75ef 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2448,6 +2448,30 @@ def supports_url_context(model: str, custom_llm_provider: Optional[str] = None) ) +def supports_assistant_prefill( + model: str, custom_llm_provider: Optional[str] = None +) -> bool: + """ + Check if the given model supports assistant message prefill and return a boolean value. + + Anthropic removed assistant prefill starting with Claude Sonnet 4.6 / Opus 4.6 — + a prefilled assistant message returns a 400 error on those models. + https://platform.claude.com/docs/en/about-claude/models/migration-guide + + Parameters: + model (str): The model name to be checked. + custom_llm_provider (Optional[str]): The provider to be checked. + + Returns: + bool: True if the model supports assistant prefill, False otherwise (including unknown models). + """ + return _supports_factory( + model=model, + custom_llm_provider=custom_llm_provider, + key="supports_assistant_prefill", + ) + + def supports_native_streaming(model: str, custom_llm_provider: Optional[str]) -> bool: """ Check if the given model supports native streaming and return a boolean value. diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ba8b09498e8..c1a2494fbdc 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -1652,7 +1652,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1681,7 +1681,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1710,7 +1710,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1739,7 +1739,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1768,7 +1768,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -1797,7 +1797,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2470,7 +2470,7 @@ "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -10240,7 +10240,7 @@ "search_context_size_medium": 0.01 }, "supports_adaptive_thinking": true, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -34982,7 +34982,7 @@ "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -42552,7 +42552,7 @@ "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, diff --git a/tests/test_litellm/router_utils/test_fallback_event_handlers.py b/tests/test_litellm/router_utils/test_fallback_event_handlers.py new file mode 100644 index 00000000000..e65de229a5d --- /dev/null +++ b/tests/test_litellm/router_utils/test_fallback_event_handlers.py @@ -0,0 +1,95 @@ +"""Tests for mid-stream fallback continuation message building. + +Anthropic removed assistant prefill starting with Claude Sonnet 4.6 / Opus 4.6 +(a prefilled assistant message returns a 400 error), so the mid-stream fallback +must use the documented user-message continuation pattern for those models: +https://platform.claude.com/docs/en/about-claude/models/migration-guide + +Models whose registry entry says supports_assistant_prefill=true, has no value, +or is unknown keep the legacy prefill-resume behavior. +""" + +import pytest + +import litellm +from litellm.router_utils.fallback_event_handlers import ( + MID_STREAM_CONTINUATION_SYSTEM_PROMPT, + build_mid_stream_continuation_messages, +) + + +@pytest.fixture(autouse=True) +def local_model_cost_map(monkeypatch): + """Pin capability lookups to the in-repo cost map so the tests exercise this + PR's registry changes instead of the remote map.""" + original_model_cost = litellm.model_cost + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + litellm.model_cost = litellm.get_model_cost_map(url="") + litellm.get_model_info.cache_clear() + try: + yield + finally: + litellm.model_cost = original_model_cost + litellm.get_model_info.cache_clear() + +MESSAGES = [{"role": "user", "content": "Plan my trip to Tokyo"}] +PARTIAL = "Here are the best flight options I found so" + + +def _build(model_group): + return build_mid_stream_continuation_messages( + messages=MESSAGES, + generated_content=PARTIAL, + model_group=model_group, + ) + + +def _assert_legacy_prefill(result): + assert len(result) == 3 + assert result[0] == MESSAGES[0] + assert result[1] == { + "role": "system", + "content": MID_STREAM_CONTINUATION_SYSTEM_PROMPT, + } + assert result[2] == { + "role": "assistant", + "content": PARTIAL, + "prefix": True, + } + + +@pytest.mark.parametrize( + "model", + [ + "claude-sonnet-4-6", + "anthropic/claude-sonnet-4-6", + "vertex_ai/claude-sonnet-4-6", + "claude-opus-4-6", + ], +) +def test_prefill_rejecting_models_get_user_continuation(model): + """Claude Sonnet 4.6+/Opus 4.6+ reject assistant prefill with a 400 — + the continuation must ride a user message and carry the partial text.""" + result = _build(model) + assert len(result) == 2 + assert result[0] == MESSAGES[0] + assert result[1]["role"] == "user" + assert PARTIAL in result[1]["content"] + assert "Continue from where you left off" in result[1]["content"] + # No prefill anywhere — the conversation must end with a user message. + assert all(m.get("prefix") is not True for m in result) + + +@pytest.mark.parametrize( + "model", + [ + "claude-3-5-sonnet-20241022", # registry: supports_assistant_prefill=true + "gpt-4", # registry entry exists, capability field absent + "definitely-not-a-real-model", # unknown model → capability lookup fails + None, # no model group available + ], +) +def test_other_models_keep_legacy_prefill_resume(model): + """Anything not explicitly marked supports_assistant_prefill=false keeps the + pre-existing prefill-resume behavior (back-compat).""" + _assert_legacy_prefill(_build(model))