From 3cf27762444bcd5a39a271da40f2ce0177e26a60 Mon Sep 17 00:00:00 2001 From: JingHao-Leon <102573344+JingHao-Leon@users.noreply.github.com> Date: Fri, 25 Sep 2026 10:15:45 +0800 Subject: [PATCH] fix(exception_mapping): recognize current OpenAI and Moonshot context window errors is_error_str_context_window_exceeded matched OpenAI's older 'This model's maximum context length is' phrasing but not the current 'exceeds the context window of this model' message, and had no pattern for Moonshot's 'exceeded model token limit: N'. Both surfaced as plain BadRequestError, so context_window_fallbacks never triggered and clients could not distinguish them from other 400s. Add both substrings to the known list and cover them in the parametrized tests. Fixes #43013 --- litellm/litellm_core_utils/exception_mapping_utils.py | 2 ++ .../litellm_core_utils/test_exception_mapping_utils.py | 7 +++++++ 2 files changed, 9 insertions(+) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 0fdfb301291..7d078b364b4 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -92,11 +92,13 @@ class ExceptionCheckers: known_exception_substrings: Final = [ "exceed context limit", "this model's maximum context length is", + "exceeds the context window of this model", # OpenAI (current wording) "string too long. expected a string with maximum length", "model's maximum context limit", "is longer than the model's context length", "input tokens exceed the configured limit", "`inputs` tokens + `max_new_tokens` must be", + "exceeded model token limit", # Moonshot "exceeds the available context size", # llama.cpp/Lemonade "exceeds the maximum number of tokens allowed", # Gemini ] diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index 9de768ea47b..115ffd2f2d1 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -64,6 +64,13 @@ context_window_test_cases = [ 'GeminiException BadRequestError - {\n "error": {\n "code": 400,\n "message": "The input token count (2800010) exceeds the maximum number of tokens allowed (1048575).",\n "status": "INVALID_ARGUMENT"\n }\n}\n', True, ), + # OpenAI current context window wording + ( + "Your input exceeds the context window of this model. Please adjust your input and try again.", + True, + ), + # Moonshot token limit format + ("Invalid request: Your request exceeded model token limit: 262144", True), # Test case insensitivity ("ERROR: THIS MODEL'S MAXIMUM CONTEXT LENGTH IS 1024.", True), # Cerebras context window error format