diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 40fdf083cf8..5d5dab6f746 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -2749,7 +2749,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2784,7 +2784,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2819,7 +2819,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2854,7 +2854,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2889,7 +2889,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2924,7 +2924,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -3712,7 +3712,7 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -15056,7 +15056,7 @@ }, "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -19314,7 +19314,7 @@ "output_cost_per_token": 2.5000010000000002e-05, "output_dbu_cost_per_token": 0.000357143, "source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_function_calling": true, "supports_legacy_thinking": true, "supports_prompt_caching": true, @@ -19529,7 +19529,7 @@ "output_cost_per_token": 1.5000020000000002e-05, "output_dbu_cost_per_token": 0.000214286, "source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_function_calling": true, "supports_legacy_thinking": true, "supports_prompt_caching": true, @@ -41833,7 +41833,7 @@ "output_cost_per_token": 1.5e-05, "output_cost_per_token_above_200k_tokens": 2.25e-05, "source": "https://openrouter.ai/api/v1/models", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_prompt_caching": true, @@ -50030,7 +50030,7 @@ "mode": "chat", "output_cost_per_token": 1.5e-05, "output_cost_per_token_batches": 7.5e-06, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -57738,7 +57738,7 @@ "mode": "chat", "output_cost_per_token": 1.5e-05, "output_cost_per_token_batches": 7.5e-06, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -78427,7 +78427,7 @@ "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 40fdf083cf8..5d5dab6f746 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -2749,7 +2749,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2784,7 +2784,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2819,7 +2819,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2854,7 +2854,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2889,7 +2889,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -2924,7 +2924,7 @@ "search_context_size_low": 0.01, "search_context_size_medium": 0.01 }, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -3712,7 +3712,7 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -15056,7 +15056,7 @@ }, "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -19314,7 +19314,7 @@ "output_cost_per_token": 2.5000010000000002e-05, "output_dbu_cost_per_token": 0.000357143, "source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_function_calling": true, "supports_legacy_thinking": true, "supports_prompt_caching": true, @@ -19529,7 +19529,7 @@ "output_cost_per_token": 1.5000020000000002e-05, "output_dbu_cost_per_token": 0.000214286, "source": "https://www.databricks.com/product/pricing/proprietary-foundation-model-serving", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_function_calling": true, "supports_legacy_thinking": true, "supports_prompt_caching": true, @@ -41833,7 +41833,7 @@ "output_cost_per_token": 1.5e-05, "output_cost_per_token_above_200k_tokens": 2.25e-05, "source": "https://openrouter.ai/api/v1/models", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_prompt_caching": true, @@ -50030,7 +50030,7 @@ "mode": "chat", "output_cost_per_token": 1.5e-05, "output_cost_per_token_batches": 7.5e-06, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -57738,7 +57738,7 @@ "mode": "chat", "output_cost_per_token": 1.5e-05, "output_cost_per_token_batches": 7.5e-06, - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, @@ -78427,7 +78427,7 @@ "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "supports_assistant_prefill": true, + "supports_assistant_prefill": false, "supports_computer_use": true, "supports_function_calling": true, "supports_pdf_input": true, diff --git a/tests/unit/router_utils/pre_call_checks/test_continuation_prefill_check.py b/tests/unit/router_utils/pre_call_checks/test_continuation_prefill_check.py index 1d6b667464e..bf3ff3c6b4f 100644 --- a/tests/unit/router_utils/pre_call_checks/test_continuation_prefill_check.py +++ b/tests/unit/router_utils/pre_call_checks/test_continuation_prefill_check.py @@ -1,5 +1,6 @@ import pytest +import litellm from litellm.router_utils.pre_call_checks.continuation_prefill_check import ( MID_STREAM_CONTINUATION_KWARG, MID_STREAM_CONTINUATION_MARKER, @@ -7,8 +8,21 @@ from litellm.router_utils.pre_call_checks.continuation_prefill_check import ( _deployment_supports_prefill, ) -PREFILL_MODEL = "anthropic/claude-3-opus-20240229" # supports_assistant_prefill: True in the cost map -NON_PREFILL_MODEL = "openai/gpt-4o" # capability absent -> treated as unsupported +PREFILL_MODEL = "anthropic/prefill-capable-test-model" +NON_PREFILL_MODEL = "openai/prefill-unknown-test-model" +UNMAPPED_MODEL = "openai/unmapped-test-model" + + +@pytest.fixture(autouse=True) +def _synthetic_cost_map_entries() -> None: + litellm.register_model( + { + PREFILL_MODEL: {"litellm_provider": "anthropic", "mode": "chat", "supports_assistant_prefill": True}, + NON_PREFILL_MODEL: {"litellm_provider": "openai", "mode": "chat"}, + }, + persist_across_reloads=False, + ) + litellm.get_model_info.cache_clear() def _deployment(model: str, dep_id: str) -> dict: @@ -96,7 +110,7 @@ async def test_filter_empties_group_when_no_prefill_capable_deployment(): """No prefill-capable deployment -> empty result, so the router advances the fallback chain and ultimately surfaces the original error.""" check = ContinuationPrefillDeploymentCheck() - deployments = [_deployment(NON_PREFILL_MODEL, "b"), _deployment("openai/gpt-4.1", "c")] + deployments = [_deployment(NON_PREFILL_MODEL, "b"), _deployment(UNMAPPED_MODEL, "c")] result = await check.async_filter_deployments( model="group",