diff --git a/litellm/constants.py b/litellm/constants.py index 21e30bef32b..9f55d2a94ef 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -51,6 +51,9 @@ SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD = int( DEFAULT_REASONING_EFFORT_DISABLE_THINKING_BUDGET = int( os.getenv("DEFAULT_REASONING_EFFORT_DISABLE_THINKING_BUDGET", 0) ) +DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET = int( + os.getenv("DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET", 128) +) DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET = int( os.getenv("DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET", 1024) ) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 25ae0269ab3..ad6b3dcaeb4 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -24,6 +24,55 @@ from ..exceptions import ( ) +def _is_operational_404(original_exception) -> bool: + """ + Determine if a 404 status code represents an operational issue rather than a missing model. + + Args: + original_exception: The exception with status_code 404 + + Returns: + True if this is an operational issue (rate limiting, cooldowns, etc.) + False if this is actually a missing model + """ + try: + # Import here to avoid circular imports + from litellm.types.router import RouterErrors + + # Check for known operational error patterns + error_message = str(original_exception).lower() + + # Check for router-specific operational errors + operational_patterns = [ + RouterErrors.no_deployments_available.value.lower(), + "no deployments available", + "no healthy deployment available", + "no healthy deployments available", + "deployment over user-defined ratelimit", + "crossed budget", + "cooldown", + "rate limit exceeded", + "too many requests" + ] + + for pattern in operational_patterns: + if pattern in error_message: + return True + + # Check if this is a RouterRateLimitError (which indicates operational issues) + if hasattr(original_exception, '__class__'): + exception_class_name = original_exception.__class__.__name__ + if "RouterRateLimitError" in exception_class_name: + return True + + return False + + except Exception: + # If we can't determine, default to treating it as a missing model + # This is safer than potentially hiding real model not found errors + return False + + class ExceptionCheckers: """ Helper class for checking various error conditions in exception strings. @@ -462,13 +511,26 @@ def exception_type( # type: ignore # noqa: PLR0915 ) elif original_exception.status_code == 404: exception_mapping_worked = True - raise NotFoundError( - message=f"NotFoundError: {exception_provider} - {message}", - model=model, - llm_provider=custom_llm_provider, - response=getattr(original_exception, "response", None), - litellm_debug_info=extra_information, - ) + # Check if this is actually a "model not found" vs operational issue + if _is_operational_404(original_exception): + # This is operational (rate limiting, cooldowns), not a missing model + # The proxy will map this to 429 status code, which is correct + raise litellm.ServiceUnavailableError( + message=f"ServiceUnavailableError: {exception_provider} - {message}", + model=model, + llm_provider=custom_llm_provider, + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) + else: + # This is actually a missing model + raise NotFoundError( + message=f"NotFoundError: {exception_provider} - {message}", + model=model, + llm_provider=custom_llm_provider, + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) elif original_exception.status_code == 408: exception_mapping_worked = True raise Timeout( diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index 37470a6ee09..4da99204165 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -30,6 +30,7 @@ from litellm.constants import ( DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET, ) from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.llms.custom_httpx.http_handler import ( @@ -423,7 +424,12 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): def _map_reasoning_effort_to_thinking_budget( reasoning_effort: str, ) -> GeminiThinkingConfig: - if reasoning_effort == "low": + if reasoning_effort == "minimal": + return { + "thinkingBudget": DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET, + "includeThoughts": True, + } + elif reasoning_effort == "low": return { "thinkingBudget": DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, "includeThoughts": True,