fix(exception_mapping): recognize current OpenAI and Moonshot context window errors

is_error_str_context_window_exceeded matched OpenAI's older 'This model's
maximum context length is' phrasing but not the current 'exceeds the context
window of this model' message, and had no pattern for Moonshot's 'exceeded
model token limit: N'. Both surfaced as plain BadRequestError, so
context_window_fallbacks never triggered and clients could not distinguish
them from other 400s.

Add both substrings to the known list and cover them in the parametrized
tests.

Fixes #43013
This commit is contained in:
JingHao-Leon 2026-09-25 10:15:45 +08:00
parent 118ce3cc91
commit 3cf2776244
2 changed files with 9 additions and 0 deletions

View file

@ -92,11 +92,13 @@ class ExceptionCheckers:
known_exception_substrings: Final = [
"exceed context limit",
"this model's maximum context length is",
"exceeds the context window of this model", # OpenAI (current wording)
"string too long. expected a string with maximum length",
"model's maximum context limit",
"is longer than the model's context length",
"input tokens exceed the configured limit",
"`inputs` tokens + `max_new_tokens` must be",
"exceeded model token limit", # Moonshot
"exceeds the available context size", # llama.cpp/Lemonade
"exceeds the maximum number of tokens allowed", # Gemini
]

View file

@ -64,6 +64,13 @@ context_window_test_cases = [
'GeminiException BadRequestError - {\n "error": {\n "code": 400,\n "message": "The input token count (2800010) exceeds the maximum number of tokens allowed (1048575).",\n "status": "INVALID_ARGUMENT"\n }\n}\n',
True,
),
# OpenAI current context window wording
(
"Your input exceeds the context window of this model. Please adjust your input and try again.",
True,
),
# Moonshot token limit format
("Invalid request: Your request exceeded model token limit: 262144", True),
# Test case insensitivity
("ERROR: THIS MODEL'S MAXIMUM CONTEXT LENGTH IS 1024.", True),
# Cerebras context window error format