feat(proxy): allow custom client-facing budget exceeded message

This commit is contained in:
jacksonriding 2026-08-15 12:01:33 +10:00
parent 1a183efaa1
commit 832eb193ab
9 changed files with 79 additions and 4 deletions

View file

@ -394,6 +394,7 @@ budget_duration: Optional[str] = (
)
default_soft_budget: float = DEFAULT_SOFT_BUDGET # by default all litellm proxy keys have a soft budget of 50.0
budget_exceeded_throttle_percentage: Optional[float] = None
budget_exceeded_error_message: Optional[str] = None
forward_traceparent_to_llm_provider: bool = False

View file

@ -1570,6 +1570,7 @@ LITELLM_SETTINGS_SAFE_DB_OVERRIDES: Final = [
"cost_discount_config",
"cost_margin_config",
"budget_exceeded_throttle_percentage",
"budget_exceeded_error_message",
# Every field editable from the Admin UI (proxy_server._GENERAL_SETTINGS_UI_LITELLM_FIELDS)
# must be listed here so a DB write from one worker overrides the live litellm attribute on
# the others when config reloads; otherwise peer workers stay on their startup value.

View file

@ -985,6 +985,12 @@ class BudgetExceededError(Exception):
self.message = message
super().__init__(message)
@property
def client_facing_message(self) -> str:
import litellm
return litellm.budget_exceeded_error_message or self.message
## DEPRECATED ##
class InvalidRequestError(openai.BadRequestError):

View file

@ -1135,7 +1135,7 @@ class MCPRequestHandler:
except (HTTPException, ProxyException):
raise
except litellm.BudgetExceededError as e:
raise HTTPException(status_code=getattr(e, "status_code", 429), detail=str(e)) from None
raise HTTPException(status_code=getattr(e, "status_code", 429), detail=e.client_facing_message) from None
except Exception as e: # noqa: BLE001 # untyped gate failure: retryable 503 for a DB outage, else fail closed 401
MCPRequestHandler._raise_503_if_db_unavailable(e)
raise HTTPException(status_code=401, detail="Invalid or expired credential") from None

View file

@ -145,7 +145,7 @@ class UserAPIKeyAuthExceptionHandler:
if isinstance(e, litellm.BudgetExceededError):
raise ProxyException(
message=e.message,
message=e.client_facing_message,
type=ProxyErrorTypes.budget_exceeded,
param=None,
code=getattr(e, "status_code", status.HTTP_429_TOO_MANY_REQUESTS),

View file

@ -40,7 +40,7 @@ from litellm.litellm_core_utils.llm_response_utils.get_headers import (
get_response_headers,
)
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.proxy._types import ProxyException, UserAPIKeyAuth
from litellm.proxy._types import ProxyErrorTypes, ProxyException, UserAPIKeyAuth
from litellm.proxy.auth.auth_checks import can_key_call_resolved_model
from litellm.proxy.auth.auth_utils import check_response_size_is_safe
from litellm.proxy.common_utils.callback_utils import (
@ -2881,6 +2881,15 @@ class ProxyBaseLLMRequestProcessing:
code=status.HTTP_400_BAD_REQUEST,
headers=safe_headers,
)
if isinstance(e, litellm.BudgetExceededError):
raise ProxyException(
message=e.client_facing_message,
type=ProxyErrorTypes.budget_exceeded,
param=None,
code=e.status_code,
headers=safe_headers,
)
# Extract status_code from the exception if it carries one.
# Provider exceptions (NotFoundError, BadRequestError, GeminiError,
# VertexAIError, etc.) all have a status_code attribute reflecting

View file

@ -162,7 +162,7 @@ async def _authorize_models_this_test_can_call(
)
except BudgetExceededError as e:
raise ProxyException(
message=e.message,
message=e.client_facing_message,
type=ProxyErrorTypes.budget_exceeded,
param=None,
code=status.HTTP_400_BAD_REQUEST,

View file

@ -30,6 +30,7 @@ sys.path.insert(
0, os.path.abspath("../../..")
) # Adds the parent directory to the system path
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import ProxyErrorTypes, ProxyException, UserAPIKeyAuth
from litellm.proxy.auth.auth_exception_handler import UserAPIKeyAuthExceptionHandler
@ -356,6 +357,41 @@ async def test_handle_authentication_error_budget_exceeded():
assert int(exc_info.value.code) == status.HTTP_429_TOO_MANY_REQUESTS
@pytest.mark.asyncio
async def test_handle_authentication_error_budget_exceeded_custom_message(monkeypatch):
from litellm.exceptions import BudgetExceededError
monkeypatch.setattr(
litellm,
"budget_exceeded_error_message",
"Your AI usage allowance has been reached. Please contact the AI team.",
)
budget_error = BudgetExceededError(
message="Budget has been exceeded! Current cost: 10.51, Max budget: 10.00",
current_cost=10.51,
max_budget=10.00,
)
with pytest.raises(ProxyException) as exc_info:
await UserAPIKeyAuthExceptionHandler()._handle_authentication_error(
budget_error,
MagicMock(),
{},
"/v1/chat/completions",
None,
"sk-test",
)
assert (
exc_info.value.message
== "Your AI usage allowance has been reached. Please contact the AI team."
)
assert exc_info.value.type == ProxyErrorTypes.budget_exceeded
assert int(exc_info.value.code) == status.HTTP_429_TOO_MANY_REQUESTS
assert "10.51" in budget_error.message
@pytest.mark.asyncio
async def test_route_passed_to_post_call_failure_hook():
"""

View file

@ -3190,6 +3190,28 @@ class TestHandleLLMApiExceptionRetryAfter:
assert proxy_exc.headers["retry-after"] == "43"
assert proxy_exc.headers["x-custom"] == "1"
async def test_handle_llm_api_exception_budget_exceeded_uses_custom_message(
self, monkeypatch
):
monkeypatch.setattr(
litellm, "budget_exceeded_error_message", "Allowance reached, contact the AI team."
)
exc = litellm.BudgetExceededError(current_cost=10.51, max_budget=10.0)
proxy_exc = await self._invoke(exc)
assert proxy_exc.message == "Allowance reached, contact the AI team."
assert proxy_exc.type == "budget_exceeded"
assert proxy_exc.code == "429"
async def test_handle_llm_api_exception_budget_exceeded_defaults_to_detail(self):
exc = litellm.BudgetExceededError(current_cost=10.51, max_budget=10.0)
proxy_exc = await self._invoke(exc)
assert "10.51" in proxy_exc.message
assert proxy_exc.type == "budget_exceeded"
class TestHandleLLMApiExceptionFramingHeaders:
"""HTTP-framing headers on the provider exception must be stripped before the