fix(aiand): address review findings on provider parity

- aiand errors that match none of the aiand codes now fall through to
  the openai-compatible mapper, so vLLM context-window strings keep
  their ContextWindowExceededError classification instead of a generic
  BadRequestError
- 402 insufficient_credits maps to PermissionDeniedError instead of
  RateLimitError, so the router no longer retries an exhausted balance
- sync script treats reasoning_effort_levels as spec-owned in both
  directions: a catalog-dropped effort option proposes [] instead of
  preserving stale levels
- tests assert litellm-owned invariants (enum membership, reasoning
  consistency, backup/runtime parity) instead of pinning vendor facts
  like exact effort levels or a 13-model count
This commit is contained in:
fenil modi 2026-10-03 14:14:12 +00:00
parent c77aa2f882
commit bb22beacec
5 changed files with 62 additions and 26 deletions

View file

@ -1839,11 +1839,11 @@ def _map_aiand_exception(
or "insufficient_credits" in error_str
or "billing_error" in error_str
):
raise RateLimitError(
raise PermissionDeniedError(
message=f"{exception_provider} - {message}",
llm_provider="aiand",
model=model,
response=getattr(original_exception, "response", None),
response=_response_or_stub(original_exception, status_code=403),
litellm_debug_info=extra_information,
)
elif status_code == 401 and (error_payload.get("code") == "invalid_api_key" or "invalid_api_key" in error_str):
@ -2538,7 +2538,7 @@ def exception_type(
exception_provider=exception_provider,
extra_information=extra_information,
)
elif (
if (
custom_llm_provider == "openai"
or custom_llm_provider == "text-completion-openai"
or custom_llm_provider == "custom_openai"

View file

@ -136,8 +136,7 @@ def _spec_fields(model: SpecModel) -> RegistryEntry:
"source": SOURCE_URL,
"supported_endpoints": list(SUPPORTED_ENDPOINTS),
}
if effort_levels:
fields["reasoning_effort_levels"] = list(effort_levels)
fields["reasoning_effort_levels"] = list(effort_levels)
return fields

View file

@ -1064,7 +1064,7 @@ AIAND_INSUFFICIENT_CREDITS_MESSAGE = (
},
],
)
def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exception_mapping):
def test_an_aiand_402_billing_error_is_a_permission_denied_error(error_body, quiet_exception_mapping):
from litellm.llms.base_llm.chat.transformation import BaseLLMException
original_exception = BaseLLMException(
@ -1072,20 +1072,20 @@ def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exce
message=json.dumps({"error": error_body}),
)
with pytest.raises(litellm.RateLimitError) as raised:
with pytest.raises(litellm.PermissionDeniedError) as raised:
exception_type(
model="test-model",
original_exception=original_exception,
custom_llm_provider="aiand",
)
assert raised.value.status_code == 429
assert raised.value.status_code == 402
assert raised.value.llm_provider == "aiand"
assert raised.value.model == "test-model"
assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_body['message']}"
assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_body['message']}"
def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_exception_mapping):
def test_an_aiand_402_from_the_response_body_is_a_permission_denied_error(quiet_exception_mapping):
from litellm.llms.base_llm.chat.transformation import BaseLLMException
original_exception = BaseLLMException(
@ -1101,7 +1101,7 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti
},
)
with pytest.raises(litellm.RateLimitError) as raised:
with pytest.raises(litellm.PermissionDeniedError) as raised:
exception_type(
model="test-model",
original_exception=original_exception,
@ -1109,10 +1109,12 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti
)
assert raised.value.llm_provider == "aiand"
assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
assert (
raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
)
def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_exception_mapping):
def test_an_aiand_402_from_an_unwrapped_body_is_a_permission_denied_error(quiet_exception_mapping):
from litellm.llms.base_llm.chat.transformation import BaseLLMException
original_exception = BaseLLMException(
@ -1130,19 +1132,21 @@ def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_excepti
},
)
with pytest.raises(litellm.RateLimitError) as raised:
with pytest.raises(litellm.PermissionDeniedError) as raised:
exception_type(
model="test-model",
original_exception=original_exception,
custom_llm_provider="aiand",
)
assert raised.value.status_code == 429
assert raised.value.status_code == 402
assert raised.value.llm_provider == "aiand"
assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
assert (
raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
)
def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exception_mapping):
def test_an_aiand_402_from_a_non_json_error_str_is_a_permission_denied_error(quiet_exception_mapping):
from litellm.llms.base_llm.chat.transformation import BaseLLMException
error_str = (
@ -1152,16 +1156,16 @@ def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exce
)
original_exception = BaseLLMException(status_code=402, message=error_str)
with pytest.raises(litellm.RateLimitError) as raised:
with pytest.raises(litellm.PermissionDeniedError) as raised:
exception_type(
model="test-model",
original_exception=original_exception,
custom_llm_provider="aiand",
)
assert raised.value.status_code == 429
assert raised.value.status_code == 402
assert raised.value.llm_provider == "aiand"
assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_str}"
assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_str}"
def test_an_aiand_401_invalid_api_key_is_an_authentication_error(quiet_exception_mapping):
@ -1224,6 +1228,25 @@ def test_an_aiand_404_model_not_found_is_a_not_found_error(quiet_exception_mappi
assert raised.value.message == "litellm.NotFoundError: AiandException - Model not found"
def test_an_aiand_context_window_error_is_a_context_window_exceeded_error(quiet_exception_mapping):
from litellm.llms.base_llm.chat.transformation import BaseLLMException
original_exception = BaseLLMException(status_code=400, message=CONTEXT_WINDOW_MESSAGE)
with pytest.raises(litellm.ContextWindowExceededError) as raised:
exception_type(
model="test-model",
original_exception=original_exception,
custom_llm_provider="aiand",
)
assert type(raised.value) is litellm.ContextWindowExceededError
assert raised.value.status_code == 400
assert raised.value.llm_provider == "aiand"
assert raised.value.model == "test-model"
assert f"ContextWindowExceededError: AiandException - {CONTEXT_WINDOW_MESSAGE}" in raised.value.message
def test_an_unknown_aiand_error_falls_through_without_raising():
from litellm.llms.base_llm.chat.transformation import BaseLLMException

View file

@ -96,14 +96,16 @@ def test_aiand_model_cost_and_capabilities(model: str) -> None:
assert litellm.supports_vision(model) is model_info["supports_vision"]
def test_aiand_entries_declare_reasoning_effort_levels() -> None:
def test_aiand_reasoning_effort_levels_are_valid() -> None:
known_efforts: Final = frozenset(get_args(REASONING_EFFORT))
for model in AIAND_MODELS:
levels = litellm.get_model_info(model)["reasoning_effort_levels"]
assert levels, f"{model} declares no reasoning_effort_levels"
model_info = litellm.get_model_info(model)
levels = model_info.get("reasoning_effort_levels", [])
assert set(levels) <= known_efforts, f"{model} declares unknown reasoning efforts"
flash = litellm.get_model_info("aiand/deepseek-ai/deepseek-v4.1-flash")
assert set(flash["reasoning_effort_levels"]) == {"none", "high", "max"}
if levels:
assert model_info["supports_reasoning"] is True, (
f"{model} declares reasoning efforts without supports_reasoning"
)
def test_aiand_backup_registry_mirrors_cost_map() -> None:
@ -122,7 +124,7 @@ def test_aiand_models_listed_by_provider(monkeypatch: pytest.MonkeyPatch) -> Non
package_root = Path(litellm.__file__).parent
backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text())
aiand_keys = {name for name in backup if name.startswith("aiand/")}
assert len(aiand_keys) == 13
assert aiand_keys
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))

View file

@ -106,6 +106,18 @@ def test_updated_price_is_detected_and_rendered() -> None:
assert outcome.has_changes is True
def test_dropped_effort_option_is_reported_and_cleared() -> None:
offered = _model(reasoning_options=[{"type": "effort", "values": ["low", "high"]}])
baseline = sync.compute_sync({}, sync.load_spec(_spec_json(offered)))
outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(_model())))
assert outcome.added == ()
assert len(outcome.updated) == 1
assert outcome.updated[0].startswith("aiand/acme/chat-1:")
assert "reasoning_effort_levels" in outcome.updated[0]
assert outcome.cost_map["aiand/acme/chat-1"]["reasoning_effort_levels"] == []
assert outcome.has_changes is True
def test_removed_model_is_stamped_and_counted_as_updated(
monkeypatch: pytest.MonkeyPatch,
) -> None: