mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix(aiand): address review findings on provider parity
- aiand errors that match none of the aiand codes now fall through to the openai-compatible mapper, so vLLM context-window strings keep their ContextWindowExceededError classification instead of a generic BadRequestError - 402 insufficient_credits maps to PermissionDeniedError instead of RateLimitError, so the router no longer retries an exhausted balance - sync script treats reasoning_effort_levels as spec-owned in both directions: a catalog-dropped effort option proposes [] instead of preserving stale levels - tests assert litellm-owned invariants (enum membership, reasoning consistency, backup/runtime parity) instead of pinning vendor facts like exact effort levels or a 13-model count
This commit is contained in:
parent
c77aa2f882
commit
bb22beacec
5 changed files with 62 additions and 26 deletions
|
|
@ -1839,11 +1839,11 @@ def _map_aiand_exception(
|
|||
or "insufficient_credits" in error_str
|
||||
or "billing_error" in error_str
|
||||
):
|
||||
raise RateLimitError(
|
||||
raise PermissionDeniedError(
|
||||
message=f"{exception_provider} - {message}",
|
||||
llm_provider="aiand",
|
||||
model=model,
|
||||
response=getattr(original_exception, "response", None),
|
||||
response=_response_or_stub(original_exception, status_code=403),
|
||||
litellm_debug_info=extra_information,
|
||||
)
|
||||
elif status_code == 401 and (error_payload.get("code") == "invalid_api_key" or "invalid_api_key" in error_str):
|
||||
|
|
@ -2538,7 +2538,7 @@ def exception_type(
|
|||
exception_provider=exception_provider,
|
||||
extra_information=extra_information,
|
||||
)
|
||||
elif (
|
||||
if (
|
||||
custom_llm_provider == "openai"
|
||||
or custom_llm_provider == "text-completion-openai"
|
||||
or custom_llm_provider == "custom_openai"
|
||||
|
|
|
|||
|
|
@ -136,8 +136,7 @@ def _spec_fields(model: SpecModel) -> RegistryEntry:
|
|||
"source": SOURCE_URL,
|
||||
"supported_endpoints": list(SUPPORTED_ENDPOINTS),
|
||||
}
|
||||
if effort_levels:
|
||||
fields["reasoning_effort_levels"] = list(effort_levels)
|
||||
fields["reasoning_effort_levels"] = list(effort_levels)
|
||||
return fields
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1064,7 +1064,7 @@ AIAND_INSUFFICIENT_CREDITS_MESSAGE = (
|
|||
},
|
||||
],
|
||||
)
|
||||
def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exception_mapping):
|
||||
def test_an_aiand_402_billing_error_is_a_permission_denied_error(error_body, quiet_exception_mapping):
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
original_exception = BaseLLMException(
|
||||
|
|
@ -1072,20 +1072,20 @@ def test_an_aiand_402_billing_error_is_a_rate_limit_error(error_body, quiet_exce
|
|||
message=json.dumps({"error": error_body}),
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.RateLimitError) as raised:
|
||||
with pytest.raises(litellm.PermissionDeniedError) as raised:
|
||||
exception_type(
|
||||
model="test-model",
|
||||
original_exception=original_exception,
|
||||
custom_llm_provider="aiand",
|
||||
)
|
||||
|
||||
assert raised.value.status_code == 429
|
||||
assert raised.value.status_code == 402
|
||||
assert raised.value.llm_provider == "aiand"
|
||||
assert raised.value.model == "test-model"
|
||||
assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_body['message']}"
|
||||
assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_body['message']}"
|
||||
|
||||
|
||||
def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_exception_mapping):
|
||||
def test_an_aiand_402_from_the_response_body_is_a_permission_denied_error(quiet_exception_mapping):
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
original_exception = BaseLLMException(
|
||||
|
|
@ -1101,7 +1101,7 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti
|
|||
},
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.RateLimitError) as raised:
|
||||
with pytest.raises(litellm.PermissionDeniedError) as raised:
|
||||
exception_type(
|
||||
model="test-model",
|
||||
original_exception=original_exception,
|
||||
|
|
@ -1109,10 +1109,12 @@ def test_an_aiand_402_from_the_response_body_is_a_rate_limit_error(quiet_excepti
|
|||
)
|
||||
|
||||
assert raised.value.llm_provider == "aiand"
|
||||
assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
|
||||
assert (
|
||||
raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
|
||||
)
|
||||
|
||||
|
||||
def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_exception_mapping):
|
||||
def test_an_aiand_402_from_an_unwrapped_body_is_a_permission_denied_error(quiet_exception_mapping):
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
original_exception = BaseLLMException(
|
||||
|
|
@ -1130,19 +1132,21 @@ def test_an_aiand_402_from_an_unwrapped_body_is_a_rate_limit_error(quiet_excepti
|
|||
},
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.RateLimitError) as raised:
|
||||
with pytest.raises(litellm.PermissionDeniedError) as raised:
|
||||
exception_type(
|
||||
model="test-model",
|
||||
original_exception=original_exception,
|
||||
custom_llm_provider="aiand",
|
||||
)
|
||||
|
||||
assert raised.value.status_code == 429
|
||||
assert raised.value.status_code == 402
|
||||
assert raised.value.llm_provider == "aiand"
|
||||
assert raised.value.message == f"litellm.RateLimitError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
|
||||
assert (
|
||||
raised.value.message == f"litellm.PermissionDeniedError: AiandException - {AIAND_INSUFFICIENT_CREDITS_MESSAGE}"
|
||||
)
|
||||
|
||||
|
||||
def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exception_mapping):
|
||||
def test_an_aiand_402_from_a_non_json_error_str_is_a_permission_denied_error(quiet_exception_mapping):
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
error_str = (
|
||||
|
|
@ -1152,16 +1156,16 @@ def test_an_aiand_402_from_a_non_json_error_str_is_a_rate_limit_error(quiet_exce
|
|||
)
|
||||
original_exception = BaseLLMException(status_code=402, message=error_str)
|
||||
|
||||
with pytest.raises(litellm.RateLimitError) as raised:
|
||||
with pytest.raises(litellm.PermissionDeniedError) as raised:
|
||||
exception_type(
|
||||
model="test-model",
|
||||
original_exception=original_exception,
|
||||
custom_llm_provider="aiand",
|
||||
)
|
||||
|
||||
assert raised.value.status_code == 429
|
||||
assert raised.value.status_code == 402
|
||||
assert raised.value.llm_provider == "aiand"
|
||||
assert raised.value.message == f"litellm.RateLimitError: AiandException - {error_str}"
|
||||
assert raised.value.message == f"litellm.PermissionDeniedError: AiandException - {error_str}"
|
||||
|
||||
|
||||
def test_an_aiand_401_invalid_api_key_is_an_authentication_error(quiet_exception_mapping):
|
||||
|
|
@ -1224,6 +1228,25 @@ def test_an_aiand_404_model_not_found_is_a_not_found_error(quiet_exception_mappi
|
|||
assert raised.value.message == "litellm.NotFoundError: AiandException - Model not found"
|
||||
|
||||
|
||||
def test_an_aiand_context_window_error_is_a_context_window_exceeded_error(quiet_exception_mapping):
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
original_exception = BaseLLMException(status_code=400, message=CONTEXT_WINDOW_MESSAGE)
|
||||
|
||||
with pytest.raises(litellm.ContextWindowExceededError) as raised:
|
||||
exception_type(
|
||||
model="test-model",
|
||||
original_exception=original_exception,
|
||||
custom_llm_provider="aiand",
|
||||
)
|
||||
|
||||
assert type(raised.value) is litellm.ContextWindowExceededError
|
||||
assert raised.value.status_code == 400
|
||||
assert raised.value.llm_provider == "aiand"
|
||||
assert raised.value.model == "test-model"
|
||||
assert f"ContextWindowExceededError: AiandException - {CONTEXT_WINDOW_MESSAGE}" in raised.value.message
|
||||
|
||||
|
||||
def test_an_unknown_aiand_error_falls_through_without_raising():
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
|
|
|
|||
|
|
@ -96,14 +96,16 @@ def test_aiand_model_cost_and_capabilities(model: str) -> None:
|
|||
assert litellm.supports_vision(model) is model_info["supports_vision"]
|
||||
|
||||
|
||||
def test_aiand_entries_declare_reasoning_effort_levels() -> None:
|
||||
def test_aiand_reasoning_effort_levels_are_valid() -> None:
|
||||
known_efforts: Final = frozenset(get_args(REASONING_EFFORT))
|
||||
for model in AIAND_MODELS:
|
||||
levels = litellm.get_model_info(model)["reasoning_effort_levels"]
|
||||
assert levels, f"{model} declares no reasoning_effort_levels"
|
||||
model_info = litellm.get_model_info(model)
|
||||
levels = model_info.get("reasoning_effort_levels", [])
|
||||
assert set(levels) <= known_efforts, f"{model} declares unknown reasoning efforts"
|
||||
flash = litellm.get_model_info("aiand/deepseek-ai/deepseek-v4.1-flash")
|
||||
assert set(flash["reasoning_effort_levels"]) == {"none", "high", "max"}
|
||||
if levels:
|
||||
assert model_info["supports_reasoning"] is True, (
|
||||
f"{model} declares reasoning efforts without supports_reasoning"
|
||||
)
|
||||
|
||||
|
||||
def test_aiand_backup_registry_mirrors_cost_map() -> None:
|
||||
|
|
@ -122,7 +124,7 @@ def test_aiand_models_listed_by_provider(monkeypatch: pytest.MonkeyPatch) -> Non
|
|||
package_root = Path(litellm.__file__).parent
|
||||
backup = json.loads((package_root / "model_prices_and_context_window_backup.json").read_text())
|
||||
aiand_keys = {name for name in backup if name.startswith("aiand/")}
|
||||
assert len(aiand_keys) == 13
|
||||
assert aiand_keys
|
||||
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
|
|
|||
|
|
@ -106,6 +106,18 @@ def test_updated_price_is_detected_and_rendered() -> None:
|
|||
assert outcome.has_changes is True
|
||||
|
||||
|
||||
def test_dropped_effort_option_is_reported_and_cleared() -> None:
|
||||
offered = _model(reasoning_options=[{"type": "effort", "values": ["low", "high"]}])
|
||||
baseline = sync.compute_sync({}, sync.load_spec(_spec_json(offered)))
|
||||
outcome = sync.compute_sync(baseline.cost_map, sync.load_spec(_spec_json(_model())))
|
||||
assert outcome.added == ()
|
||||
assert len(outcome.updated) == 1
|
||||
assert outcome.updated[0].startswith("aiand/acme/chat-1:")
|
||||
assert "reasoning_effort_levels" in outcome.updated[0]
|
||||
assert outcome.cost_map["aiand/acme/chat-1"]["reasoning_effort_levels"] == []
|
||||
assert outcome.has_changes is True
|
||||
|
||||
|
||||
def test_removed_model_is_stamped_and_counted_as_updated(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue