From bb22beacec990965d908e716448718ca2e3a411b Mon Sep 17 00:00:00 2001 From: fenil modi Date: Sat, 3 Oct 2026 14:14:12 +0000 Subject: [PATCH] fix(aiand): address review findings on provider parity - aiand errors that match none of the aiand codes now fall through to the openai-compatible mapper, so vLLM context-window strings keep their ContextWindowExceededError classification instead of a generic BadRequestError - 402 insufficient_credits maps to PermissionDeniedError instead of RateLimitError, so the router no longer retries an exhausted balance - sync script treats reasoning_effort_levels as spec-owned in both directions: a catalog-dropped effort option proposes [] instead of preserving stale levels - tests assert litellm-owned invariants (enum membership, reasoning consistency, backup/runtime parity) instead of pinning vendor facts like exact effort levels or a 13-model count --- .../exception_mapping_utils.py | 6 +-- scripts/sync_aiand_models.py | 3 +- .../test_exception_mapping_utils.py | 53 +++++++++++++------ .../llms/openai_like/test_aiand_provider.py | 14 ++--- tests/unit/test_sync_aiand_models.py | 12 +++++ 5 files changed, 62 insertions(+), 26 deletions(-) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 5999bb0617d..a4948b3daa4 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -1839,11 +1839,11 @@ def _map_aiand_exception( or "insufficient_credits" in error_str or "billing_error" in error_str ): - raise RateLimitError( + raise PermissionDeniedError( message=f"{exception_provider} - {message}", llm_provider="aiand", model=model, - response=getattr(original_exception, "response", None), + response=_response_or_stub(original_exception, status_code=403), litellm_debug_info=extra_information, ) elif status_code == 401 and (error_payload.get("code") == "invalid_api_key" or "invalid_api_key" in error_str): @@ -2538,7 +2538,7 @@ def exception_type( exception_provider=exception_provider, extra_information=extra_information, ) - elif ( + if ( custom_llm_provider == "openai" or custom_llm_provider == "text-completion-openai" or custom_llm_provider == "custom_openai" diff --git a/scripts/sync_aiand_models.py b/scripts/sync_aiand_models.py index 1b2b97a4752..06e589eeaa5 100644 --- a/scripts/sync_aiand_models.py +++ b/scripts/sync_aiand_models.py @@ -136,8 +136,7 @@ def _spec_fields(model: SpecModel) -> RegistryEntry: "source": SOURCE_URL, "supported_endpoints": list(SUPPORTED_ENDPOINTS), } - if effort_levels: - fields["reasoning_effort_levels"] = list(effort_levels) + fields["reasoning_effort_levels"] = list(effort_levels) return fields diff --git a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py index c1474f3f72b..3f936e3e77c 100644 --- a/tests/unit/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/unit/litellm_core_utils/test_exception_mapping_utils.py @@ -1064,7 +1064,7 @@ AIAND_INSUFFICIENT_CREDITS_MESSAGE = ( }, ], ) -def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exception_mapping): +def test_an_aiand_402_billing_error_is_a_permission_denied_error(error_body, quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException original_exception = BaseLLMException( @@ -1072,20 +1072,20 @@ def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exce message=json.dumps({"error": error_body}), ) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, custom_llm_provider="aiand", ) - assert raised.value.status_code == 429 + assert raised.value.status_code == 402 assert raised.value.llm_provider == "aiand" assert raised.value.model == "test-model" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_body['message']}" + assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_body['message']}" -def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_exception_mapping): +def test_an_aiand_402_from_the_response_body_is_a_permission_denied_error(quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException original_exception = BaseLLMException( @@ -1101,7 +1101,7 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti }, ) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, @@ -1109,10 +1109,12 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti ) assert raised.value.llm_provider == "aiand" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + assert ( + raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + ) -def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_exception_mapping): +def test_an_aiand_402_from_an_unwrapped_body_is_a_permission_denied_error(quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException original_exception = BaseLLMException( @@ -1130,19 +1132,21 @@ def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_excepti }, ) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, custom_llm_provider="aiand", ) - assert raised.value.status_code == 429 + assert raised.value.status_code == 402 assert raised.value.llm_provider == "aiand" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + assert ( + raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}" + ) -def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exception_mapping): +def test_an_aiand_402_from_a_non_json_error_str_is_a_permission_denied_error(quiet_exception_mapping): from litellm.llms.base_llm.chat.transformation import BaseLLMException error_str = ( @@ -1152,16 +1156,16 @@ def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exce ) original_exception = BaseLLMException(status_code=402, message=error_str) - with pytest.raises(litellm.RateLimitError) as raised: + with pytest.raises(litellm.PermissionDeniedError) as raised: exception_type( model="test-model", original_exception=original_exception, custom_llm_provider="aiand", ) - assert raised.value.status_code == 429 + assert raised.value.status_code == 402 assert raised.value.llm_provider == "aiand" - assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_str}" + assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_str}" def test_an_aiand_401_invalid_api_key_is_an_authentication_error(quiet_exception_mapping): @@ -1224,6 +1228,25 @@ def test_an_aiand_404_model_not_found_is_a_not_found_error(quiet_exception_mappi assert raised.value.message == "litellm.NotFoundError: AiandException - Model not found" +def test_an_aiand_context_window_error_is_a_context_window_exceeded_error(quiet_exception_mapping): + from litellm.llms.base_llm.chat.transformation import BaseLLMException + + original_exception = BaseLLMException(status_code=400, message=CONTEXT_WINDOW_MESSAGE) + + with pytest.raises(litellm.ContextWindowExceededError) as raised: + exception_type( + model="test-model", + original_exception=original_exception, + custom_llm_provider="aiand", + ) + + assert type(raised.value) is litellm.ContextWindowExceededError + assert raised.value.status_code == 400 + assert raised.value.llm_provider == "aiand" + assert raised.value.model == "test-model" + assert f"ContextWindowExceededError: AiandException - {CONTEXT_WINDOW_MESSAGE}" in raised.value.message + + def test_an_unknown_aiand_error_falls_through_without_raising(): from litellm.llms.base_llm.chat.transformation import BaseLLMException diff --git a/tests/unit/llms/openai_like/test_aiand_provider.py b/tests/unit/llms/openai_like/test_aiand_provider.py index e91391e133f..72637b119ad 100644 --- a/tests/unit/llms/openai_like/test_aiand_provider.py +++ b/tests/unit/llms/openai_like/test_aiand_provider.py @@ -96,14 +96,16 @@ def test_aiand_model_cost_and_capabilities(model: str) -> None: assert litellm.supports_vision(model) is model_info["supports_vision"] -def test_aiand_entries_declare_reasoning_effort_levels() -> None: +def test_aiand_reasoning_effort_levels_are_valid() -> None: known_efforts: Final = frozenset(get_args(REASONING_EFFORT)) for model in AIAND_MODELS: - levels = litellm.get_model_info(model)["reasoning_effort_levels"] - assert levels, f"{model} declares no reasoning_effort_levels" + model_info = litellm.get_model_info(model) + levels = model_info.get("reasoning_effort_levels", []) assert set(levels) <= known_efforts, f"{model} declares unknown reasoning efforts" - flash = litellm.get_model_info("aiand/deepseek-ai/deepseek-v4.1-flash") - assert set(flash["reasoning_effort_levels"]) == {"none", "high", "max"} + if levels: + assert model_info["supports_reasoning"] is True, ( + f"{model} declares reasoning efforts without supports_reasoning" + ) def test_aiand_backup_registry_mirrors_cost_map() -> None: @@ -122,7 +124,7 @@ def test_aiand_models_listed_by_provider(monkeypatch: pytest.MonkeyPatch) -> Non package_root = Path(litellm.__file__).parent backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text()) aiand_keys = {name for name in backup if name.startswith("aiand/")} - assert len(aiand_keys) == 13 + assert aiand_keys monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) diff --git a/tests/unit/test_sync_aiand_models.py b/tests/unit/test_sync_aiand_models.py index 8b5e6051dde..d3ae4497d39 100644 --- a/tests/unit/test_sync_aiand_models.py +++ b/tests/unit/test_sync_aiand_models.py @@ -106,6 +106,18 @@ def test_updated_price_is_detected_and_rendered() -> None: assert outcome.has_changes is True +def test_dropped_effort_option_is_reported_and_cleared() -> None: + offered = _model(reasoning_options=[{"type": "effort", "values": ["low", "high"]}]) + baseline = sync.compute_sync({}, sync.load_spec(_spec_json(offered))) + outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(_model()))) + assert outcome.added == () + assert len(outcome.updated) == 1 + assert outcome.updated[0].startswith("aiand/acme/chat-1:") + assert "reasoning_effort_levels" in outcome.updated[0] + assert outcome.cost_map["aiand/acme/chat-1"]["reasoning_effort_levels"] == [] + assert outcome.has_changes is True + + def test_removed_model_is_stamped_and_counted_as_updated( monkeypatch: pytest.MonkeyPatch, ) -> None: