diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 0fdfb301291..7d078b364b4 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -92,11 +92,13 @@ class ExceptionCheckers: known_exception_substrings: Final = [ "exceed context limit", "this model's maximum context length is", + "exceeds the context window of this model", # OpenAI (current wording) "string too long. expected a string with maximum length", "model's maximum context limit", "is longer than the model's context length", "input tokens exceed the configured limit", "`inputs` tokens + `max_new_tokens` must be", + "exceeded model token limit", # Moonshot "exceeds the available context size", # llama.cpp/Lemonade "exceeds the maximum number of tokens allowed", # Gemini ] diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index 9de768ea47b..115ffd2f2d1 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -64,6 +64,13 @@ context_window_test_cases = [ 'GeminiException BadRequestError - {\n "error": {\n "code": 400,\n "message": "The input token count (2800010) exceeds the maximum number of tokens allowed (1048575).",\n "status": "INVALID_ARGUMENT"\n }\n}\n', True, ), + # OpenAI current context window wording + ( + "Your input exceeds the context window of this model. Please adjust your input and try again.", + True, + ), + # Moonshot token limit format + ("Invalid request: Your request exceeded model token limit: 262144", True), # Test case insensitivity ("ERROR: THIS MODEL'S MAXIMUM CONTEXT LENGTH IS 1024.", True), # Cerebras context window error format