mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
feat(proxy): soft and hard SGR limits with UI banner and Slack alerts
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
1bafdb3c93
commit
c9ff90007d
16 changed files with 833 additions and 4 deletions
|
|
@ -1490,6 +1490,7 @@ SPEND_LOG_QUEUE_SIZE_THRESHOLD: Final = int(os.getenv("SPEND_LOG_QUEUE_SIZE_THRE
|
|||
SPEND_LOG_QUEUE_POLL_INTERVAL: Final = float(os.getenv("SPEND_LOG_QUEUE_POLL_INTERVAL", 2.0))
|
||||
SPEND_COUNTER_RESEED_LOCKS_MAX_SIZE: Final = int(os.getenv("SPEND_COUNTER_RESEED_LOCKS_MAX_SIZE", 10000))
|
||||
DEFAULT_CRON_JOB_LOCK_TTL_SECONDS: Final = int(os.getenv("DEFAULT_CRON_JOB_LOCK_TTL_SECONDS", 60)) # 1 minute
|
||||
SGR_LIMIT_CHECK_INTERVAL: Final = int(os.getenv("SGR_LIMIT_CHECK_INTERVAL", "300"))
|
||||
PROXY_BUDGET_RESCHEDULER_MIN_TIME: Final = int(os.getenv("PROXY_BUDGET_RESCHEDULER_MIN_TIME", 597))
|
||||
PROXY_BATCH_POLLING_INTERVAL: Final = int(os.getenv("PROXY_BATCH_POLLING_INTERVAL", 3600))
|
||||
MAX_OBJECTS_PER_POLL_CYCLE: Final = max(1, int(os.getenv("MAX_OBJECTS_PER_POLL_CYCLE", 50)))
|
||||
|
|
|
|||
|
|
@ -40,6 +40,7 @@ from litellm.proxy._types import (
|
|||
from litellm.repositories.team_repository import TeamRepository
|
||||
from litellm.repositories.user_repository import UserRepository
|
||||
from litellm.types.integrations.slack_alerting import *
|
||||
from litellm.types.proxy.gateway_requests import SGRLimitState, SGRLimitStatus
|
||||
|
||||
from ..email_templates.templates import *
|
||||
from .batching_handler import send_to_webhook, squash_payloads
|
||||
|
|
@ -480,6 +481,49 @@ class SlackAlerting(CustomBatchLogger):
|
|||
ttl=self.alerting_args.budget_alert_ttl,
|
||||
)
|
||||
|
||||
async def sgr_limit_alert(self, status: SGRLimitStatus) -> None:
|
||||
"""
|
||||
Alert once per window per threshold crossed on the SGR allowance.
|
||||
|
||||
The window start is part of the dedup key, so a new month re-alerts even
|
||||
while the previous month's key is still cached, and the two states are
|
||||
separate keys so a deployment that crosses the soft threshold and later
|
||||
the hard one gets both.
|
||||
"""
|
||||
if self.alerting is None or self.alert_types is None:
|
||||
return
|
||||
if AlertType.sgr_limit_alerts not in self.alert_types:
|
||||
return
|
||||
if status.state is SGRLimitState.UNDER:
|
||||
return
|
||||
|
||||
cache_key: Final = f"sgr_limit_alert:{status.window_start}:{status.state.value}"
|
||||
if await self.internal_usage_cache.async_get_cache(key=cache_key) is not None:
|
||||
return
|
||||
|
||||
hard_exceeded: Final = status.state is SGRLimitState.HARD_EXCEEDED
|
||||
headline: Final = (
|
||||
"SGR limit reached"
|
||||
if hard_exceeded
|
||||
else f"SGR soft limit reached ({int(100 * status.soft_limit / status.limit)}% of the limit)"
|
||||
)
|
||||
await self.send_alert(
|
||||
message=(
|
||||
f"*{headline}*\n"
|
||||
f"Successful gateway requests this {status.window.value}: "
|
||||
f"`{status.successful_requests:,}` of `{status.limit:,}`\n"
|
||||
f"Window started `{status.window_start}` (UTC)\n"
|
||||
"Requests are still being served, this is an alert only"
|
||||
),
|
||||
level="High" if hard_exceeded else "Medium",
|
||||
alert_type=AlertType.sgr_limit_alerts,
|
||||
)
|
||||
await self.internal_usage_cache.async_set_cache(
|
||||
key=cache_key,
|
||||
value="SENT",
|
||||
ttl=self.alerting_args.budget_alert_ttl,
|
||||
)
|
||||
|
||||
async def budget_alerts(
|
||||
self,
|
||||
type: Literal[
|
||||
|
|
@ -1239,7 +1283,7 @@ Model Info:
|
|||
message: str,
|
||||
level: Literal["Low", "Medium", "High"],
|
||||
alert_type: AlertType,
|
||||
alerting_metadata: dict,
|
||||
alerting_metadata: dict | None = None,
|
||||
user_info: WebhookEvent | None = None,
|
||||
request_model: str | None = None,
|
||||
api_base: str | None = None,
|
||||
|
|
|
|||
|
|
@ -2390,6 +2390,25 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
|
|||
None,
|
||||
description="sends alerts if requests hang for 5min+",
|
||||
)
|
||||
sgr_limit: int | None = Field(
|
||||
None,
|
||||
ge=1,
|
||||
description=(
|
||||
"Limit on successful gateway requests (SGR) per window. Crossing it alerts on the admin UI and over "
|
||||
"any configured alerting integration, it does not reject traffic. Takes precedence over an "
|
||||
"enterprise license's `max_sgr`"
|
||||
),
|
||||
)
|
||||
sgr_soft_limit_percent: float | None = Field(
|
||||
None,
|
||||
gt=0,
|
||||
le=1,
|
||||
description="Fraction of `sgr_limit` at which the soft alert fires. Defaults to 0.8",
|
||||
)
|
||||
sgr_limit_window: Literal["month", "year"] | None = Field(
|
||||
None,
|
||||
description="Period `sgr_limit` is counted over, calendar aligned in UTC. Defaults to `month`",
|
||||
)
|
||||
ui_access_mode: Literal["admin_only", "all"] | None = Field("all", description="Control access to the Proxy UI")
|
||||
allowed_routes: list | None = Field(None, description="Proxy API Endpoints you want users to be able to access")
|
||||
reject_clientside_metadata_tags: bool | None = Field(
|
||||
|
|
@ -4691,6 +4710,7 @@ class EnterpriseLicenseData(TypedDict, total=False):
|
|||
allowed_features: list[str]
|
||||
max_users: int
|
||||
max_teams: int
|
||||
max_sgr: int
|
||||
|
||||
|
||||
class ResponseLiteLLM_ManagedVectorStore(TypedDict, total=False):
|
||||
|
|
|
|||
184
litellm/proxy/db/gateway_request_limits.py
Normal file
184
litellm/proxy/db/gateway_request_limits.py
Normal file
|
|
@ -0,0 +1,184 @@
|
|||
"""
|
||||
Soft and hard limits on SGR (successful gateway requests).
|
||||
|
||||
An allowance can come from two places. An enterprise license may carry
|
||||
``max_sgr``, which makes the contracted volume visible to the deployment
|
||||
without LiteLLM having to receive any telemetry back. ``general_settings`` can
|
||||
also set it directly, which is what lets a customer self-serve a threshold
|
||||
lower than their contract, and which wins when both are present.
|
||||
|
||||
Crossing a threshold alerts, it does not reject: this reads the same
|
||||
``LiteLLM_DailyGatewayRequests`` rollup the admin UI reads, on a scheduler, so
|
||||
it is never on the request path and cannot fail a request.
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping
|
||||
from datetime import datetime, timezone
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Final, assert_never
|
||||
|
||||
from pydantic import BaseModel, Field, TypeAdapter
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.types.proxy.gateway_requests import (
|
||||
SGRLimitConfig,
|
||||
SGRLimitState,
|
||||
SGRLimitStatus,
|
||||
SGRLimitWindow,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.proxy._types import EnterpriseLicenseData
|
||||
from litellm.proxy.utils import PrismaClient, ProxyLogging
|
||||
|
||||
DEFAULT_SGR_SOFT_LIMIT_PERCENT: Final = 0.8
|
||||
|
||||
|
||||
class SGRLimitSettings(BaseModel):
|
||||
"""The `general_settings` keys that configure an SGR allowance."""
|
||||
|
||||
sgr_limit: int | None = Field(
|
||||
default=None,
|
||||
ge=1,
|
||||
description="Hard limit on successful gateway requests per window. Overrides a license's max_sgr",
|
||||
)
|
||||
sgr_soft_limit_percent: float = Field(
|
||||
default=DEFAULT_SGR_SOFT_LIMIT_PERCENT,
|
||||
gt=0,
|
||||
le=1,
|
||||
description="Fraction of the hard limit at which the soft alert fires",
|
||||
)
|
||||
sgr_limit_window: SGRLimitWindow = Field(
|
||||
default=SGRLimitWindow.MONTH,
|
||||
description="Period the limit is counted over, calendar aligned in UTC",
|
||||
)
|
||||
|
||||
|
||||
_SETTINGS_ADAPTER: Final = TypeAdapter(SGRLimitSettings)
|
||||
|
||||
|
||||
def _license_sgr_limit(license_data: "EnterpriseLicenseData | None") -> int | None:
|
||||
if license_data is None:
|
||||
return None
|
||||
max_sgr: Final = license_data.get("max_sgr")
|
||||
return max_sgr if isinstance(max_sgr, int) and max_sgr > 0 else None
|
||||
|
||||
|
||||
def resolve_sgr_limit(
|
||||
*,
|
||||
general_settings: Mapping[str, object],
|
||||
license_data: "EnterpriseLicenseData | None",
|
||||
) -> SGRLimitConfig | None:
|
||||
"""The allowance in force, or None when the deployment has not been given one."""
|
||||
try:
|
||||
settings: Final = _SETTINGS_ADAPTER.validate_python(
|
||||
MappingProxyType(
|
||||
{
|
||||
key: value
|
||||
for key, value in general_settings.items()
|
||||
if key in SGRLimitSettings.model_fields and value is not None
|
||||
}
|
||||
)
|
||||
)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.warning(
|
||||
"SGR limit - ignoring invalid sgr_limit settings in general_settings", exc_info=True
|
||||
)
|
||||
return None
|
||||
|
||||
limit: Final = settings.sgr_limit if settings.sgr_limit is not None else _license_sgr_limit(license_data)
|
||||
if limit is None:
|
||||
return None
|
||||
|
||||
return SGRLimitConfig(
|
||||
limit=limit,
|
||||
soft_limit=max(1, int(limit * settings.sgr_soft_limit_percent)),
|
||||
window=settings.sgr_limit_window,
|
||||
)
|
||||
|
||||
|
||||
def sgr_window_start(window: SGRLimitWindow, now: datetime) -> str:
|
||||
"""First day of the current window, as the YYYY-MM-DD the rollup is keyed by."""
|
||||
match window:
|
||||
case SGRLimitWindow.MONTH:
|
||||
return now.strftime("%Y-%m-01")
|
||||
case SGRLimitWindow.YEAR:
|
||||
return now.strftime("%Y-01-01")
|
||||
case _:
|
||||
assert_never(window)
|
||||
|
||||
|
||||
def evaluate_sgr_limit(config: SGRLimitConfig, successful_requests: int) -> SGRLimitState:
|
||||
if successful_requests >= config.limit:
|
||||
return SGRLimitState.HARD_EXCEEDED
|
||||
if successful_requests >= config.soft_limit:
|
||||
return SGRLimitState.SOFT_EXCEEDED
|
||||
return SGRLimitState.UNDER
|
||||
|
||||
|
||||
_WINDOW_TOTAL_SQL: Final = """
|
||||
SELECT COALESCE(SUM(successful_requests), 0)::bigint AS successful_requests
|
||||
FROM "LiteLLM_DailyGatewayRequests"
|
||||
WHERE date >= $1
|
||||
"""
|
||||
|
||||
|
||||
class _WindowTotalRow(BaseModel):
|
||||
successful_requests: int
|
||||
|
||||
|
||||
_ROWS_ADAPTER: Final = TypeAdapter(tuple[_WindowTotalRow, ...])
|
||||
|
||||
|
||||
async def get_sgr_limit_status(
|
||||
*,
|
||||
prisma_client: "PrismaClient",
|
||||
config: SGRLimitConfig,
|
||||
now: datetime,
|
||||
) -> SGRLimitStatus:
|
||||
"""Sum the window's successful requests and place them against the allowance."""
|
||||
window_start: Final = sgr_window_start(config.window, now)
|
||||
raw_rows: Final = await prisma_client.db.query_raw( # pyright: ignore[reportAny] # untyped prisma client
|
||||
_WINDOW_TOTAL_SQL,
|
||||
window_start,
|
||||
)
|
||||
rows: Final = _ROWS_ADAPTER.validate_python(raw_rows or ())
|
||||
successful_requests: Final = rows[0].successful_requests if rows else 0
|
||||
|
||||
return SGRLimitStatus(
|
||||
limit=config.limit,
|
||||
soft_limit=config.soft_limit,
|
||||
window=config.window,
|
||||
window_start=window_start,
|
||||
successful_requests=successful_requests,
|
||||
state=evaluate_sgr_limit(config, successful_requests),
|
||||
)
|
||||
|
||||
|
||||
async def check_sgr_limit(
|
||||
prisma_client: "PrismaClient",
|
||||
proxy_logging_obj: "ProxyLogging",
|
||||
) -> None:
|
||||
"""
|
||||
Scheduler entrypoint. Never raises: a limit check must not kill the job.
|
||||
|
||||
Config is resolved on every run rather than captured once, so a license or
|
||||
a `general_settings` value that arrives after startup is picked up.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import _license_check, general_settings
|
||||
|
||||
try:
|
||||
config: Final = resolve_sgr_limit(
|
||||
general_settings=general_settings,
|
||||
license_data=_license_check.airgapped_license_data,
|
||||
)
|
||||
if config is None:
|
||||
return
|
||||
status: Final = await get_sgr_limit_status(
|
||||
prisma_client=prisma_client,
|
||||
config=config,
|
||||
now=datetime.now(timezone.utc),
|
||||
)
|
||||
await proxy_logging_obj.slack_alerting_instance.sgr_limit_alert(status=status)
|
||||
except Exception: # noqa: BLE001 # a failed check must not stop the scheduler
|
||||
verbose_proxy_logger.warning("SGR limit - check failed, retrying on the next run", exc_info=True)
|
||||
|
|
@ -13,7 +13,7 @@ and the endpoint is restricted to proxy admin roles.
|
|||
|
||||
from collections.abc import Sequence
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Annotated, Final
|
||||
from typing import TYPE_CHECKING, Annotated, Final
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException, Query
|
||||
from pydantic import BaseModel, TypeAdapter
|
||||
|
|
@ -21,12 +21,17 @@ from pydantic import BaseModel, TypeAdapter
|
|||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.proxy._types import CommonProxyErrors, LitellmUserRoles, UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.db.gateway_request_limits import get_sgr_limit_status, resolve_sgr_limit
|
||||
from litellm.types.proxy.gateway_requests import (
|
||||
GatewayRequestActivityResponse,
|
||||
GatewayRequestBreakdownEntry,
|
||||
GatewayRequestDailyEntry,
|
||||
SGRLimitStatus,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.proxy.utils import PrismaClient
|
||||
|
||||
router: Final = APIRouter()
|
||||
|
||||
_DEFAULT_LOOKBACK_DAYS: Final = 30
|
||||
|
|
@ -57,6 +62,28 @@ class _AggregateRow(BaseModel):
|
|||
_ROWS_ADAPTER: Final = TypeAdapter(tuple[_AggregateRow, ...])
|
||||
|
||||
|
||||
async def _sgr_limit_status(prisma_client: "PrismaClient") -> SGRLimitStatus | None:
|
||||
"""
|
||||
Standing against the SGR allowance, or None when none is configured.
|
||||
|
||||
Read at request time rather than at import: the license and the YAML
|
||||
`general_settings` both land after this module is imported.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import _license_check, general_settings
|
||||
|
||||
config: Final = resolve_sgr_limit(
|
||||
general_settings=general_settings,
|
||||
license_data=_license_check.airgapped_license_data,
|
||||
)
|
||||
if config is None:
|
||||
return None
|
||||
try:
|
||||
return await get_sgr_limit_status(prisma_client=prisma_client, config=config, now=datetime.now(timezone.utc))
|
||||
except Exception: # noqa: BLE001 # the activity numbers are worth serving even when the limit read fails
|
||||
verbose_proxy_logger.warning("SGR limit - could not read the limit status", exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def _default_range() -> tuple[str, str]:
|
||||
end: Final = datetime.now(timezone.utc)
|
||||
start: Final = end - timedelta(days=_DEFAULT_LOOKBACK_DAYS)
|
||||
|
|
@ -136,4 +163,5 @@ async def get_gateway_daily_activity(
|
|||
total_failed_requests=sum(row.failed_requests for row in rows),
|
||||
by_date=_fold_by_date(rows),
|
||||
by_route=_fold_by_route(rows),
|
||||
sgr_limit=await _sgr_limit_status(prisma_client),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -239,6 +239,7 @@ from litellm.constants import (
|
|||
PROXY_BUDGET_RESCHEDULER_MAX_TIME,
|
||||
PROXY_BUDGET_RESCHEDULER_MIN_TIME,
|
||||
PROXY_CONFIG_RELOAD_INTERVAL_SECONDS,
|
||||
SGR_LIMIT_CHECK_INTERVAL,
|
||||
)
|
||||
from litellm.exceptions import RejectedRequestError
|
||||
from litellm.integrations.custom_guardrail import ModifyResponseException
|
||||
|
|
@ -357,6 +358,7 @@ from litellm.proxy.db.exception_handler import (
|
|||
PrismaDBExceptionHandler,
|
||||
call_with_db_reconnect_retry,
|
||||
)
|
||||
from litellm.proxy.db.gateway_request_limits import check_sgr_limit
|
||||
from litellm.proxy.db.gateway_request_tracking import (
|
||||
GatewayRequestAccumulator,
|
||||
flush_gateway_requests,
|
||||
|
|
@ -8349,6 +8351,17 @@ class ProxyStartupEvent:
|
|||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
|
||||
### ALERT ON THE GATEWAY REQUEST (SGR) LIMIT ###
|
||||
scheduler.add_job(
|
||||
check_sgr_limit,
|
||||
"interval",
|
||||
seconds=SGR_LIMIT_CHECK_INTERVAL,
|
||||
args=(prisma_client, proxy_logging_obj),
|
||||
id="check_sgr_limit_job",
|
||||
replace_existing=True,
|
||||
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
|
||||
)
|
||||
|
||||
### MONITOR SPEND LOGS QUEUE (queue-size-based job) ###
|
||||
if general_settings.get("disable_spend_logs", False) is False:
|
||||
from litellm.proxy.utils import _monitor_spend_logs_queue
|
||||
|
|
|
|||
|
|
@ -137,6 +137,7 @@ class AlertType(str, Enum):
|
|||
budget_alerts = "budget_alerts"
|
||||
spend_reports = "spend_reports"
|
||||
failed_tracking_spend = "failed_tracking_spend"
|
||||
sgr_limit_alerts = "sgr_limit_alerts"
|
||||
|
||||
# Database alerts
|
||||
db_exceptions = "db_exceptions"
|
||||
|
|
@ -180,6 +181,7 @@ DEFAULT_ALERT_TYPES: Final[list[AlertType]] = [
|
|||
AlertType.budget_alerts,
|
||||
AlertType.spend_reports,
|
||||
AlertType.failed_tracking_spend,
|
||||
AlertType.sgr_limit_alerts,
|
||||
# Database alerts
|
||||
AlertType.db_exceptions,
|
||||
# Report alerts
|
||||
|
|
|
|||
|
|
@ -2,9 +2,10 @@
|
|||
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import TypeAlias
|
||||
|
||||
from pydantic import BaseModel
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
|
|
@ -42,6 +43,39 @@ class GatewayRequestDailyEntry(BaseModel):
|
|||
failed_requests: int = 0
|
||||
|
||||
|
||||
class SGRLimitWindow(str, Enum):
|
||||
"""Period an SGR allowance is counted over. Calendar aligned and UTC."""
|
||||
|
||||
MONTH = "month"
|
||||
YEAR = "year"
|
||||
|
||||
|
||||
class SGRLimitState(str, Enum):
|
||||
SOFT_EXCEEDED = "soft_exceeded"
|
||||
HARD_EXCEEDED = "hard_exceeded"
|
||||
UNDER = "under"
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SGRLimitConfig:
|
||||
"""A resolved SGR allowance: the hard limit plus the soft alerting threshold."""
|
||||
|
||||
limit: int
|
||||
soft_limit: int
|
||||
window: SGRLimitWindow
|
||||
|
||||
|
||||
class SGRLimitStatus(BaseModel):
|
||||
"""Where the deployment sits against its SGR allowance for the current window."""
|
||||
|
||||
limit: int
|
||||
soft_limit: int
|
||||
window: SGRLimitWindow
|
||||
window_start: str
|
||||
successful_requests: int
|
||||
state: SGRLimitState
|
||||
|
||||
|
||||
class GatewayRequestActivityResponse(BaseModel):
|
||||
"""Response for GET /gateway/daily/activity."""
|
||||
|
||||
|
|
@ -49,3 +83,11 @@ class GatewayRequestActivityResponse(BaseModel):
|
|||
total_failed_requests: int = 0
|
||||
by_date: tuple[GatewayRequestDailyEntry, ...] = ()
|
||||
by_route: tuple[GatewayRequestBreakdownEntry, ...] = ()
|
||||
sgr_limit: SGRLimitStatus | None = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Standing against the configured SGR allowance, or null when no allowance is configured. "
|
||||
"The totals above cover the requested date range, this covers the limit window, "
|
||||
"so the two counts are not the same number"
|
||||
),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -9,9 +9,14 @@ from unittest.mock import ANY, AsyncMock, MagicMock, Mock, patch
|
|||
sys.path.insert(
|
||||
0, os.path.abspath("../../..")
|
||||
) # Adds the parent directory to the system-path
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.caching.dual_cache import DualCache
|
||||
from litellm.integrations.SlackAlerting.slack_alerting import SlackAlerting
|
||||
from litellm.proxy._types import CallInfo, Litellm_EntityType
|
||||
from litellm.types.integrations.slack_alerting import DEFAULT_ALERT_TYPES, AlertType
|
||||
from litellm.types.proxy.gateway_requests import SGRLimitState, SGRLimitStatus, SGRLimitWindow
|
||||
|
||||
|
||||
class TestSlackAlerting(unittest.TestCase):
|
||||
|
|
@ -245,3 +250,86 @@ class TestSlackAlerting(unittest.TestCase):
|
|||
)
|
||||
self.assertEqual(parsed_data["alerts"], [408])
|
||||
self.assertEqual(parsed_data["provider_region_id"], "vertex_aius-east1")
|
||||
|
||||
|
||||
class RecordingSlackAlerting(SlackAlerting):
|
||||
"""Records what would have been sent instead of posting to Slack."""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.sent: List[Tuple[str, str]] = []
|
||||
|
||||
async def send_alert(self, message: str, level: str, alert_type, alerting_metadata=None, **kwargs) -> None:
|
||||
self.sent.append((level, message))
|
||||
|
||||
|
||||
def _sgr_status(state: SGRLimitState, window_start: str = "2026-08-01") -> SGRLimitStatus:
|
||||
return SGRLimitStatus(
|
||||
limit=1_000_000,
|
||||
soft_limit=800_000,
|
||||
window=SGRLimitWindow.MONTH,
|
||||
window_start=window_start,
|
||||
successful_requests=900_000 if state is SGRLimitState.SOFT_EXCEEDED else 1_100_000,
|
||||
state=state,
|
||||
)
|
||||
|
||||
|
||||
def _alerting(**kwargs) -> RecordingSlackAlerting:
|
||||
return RecordingSlackAlerting(internal_usage_cache=DualCache(), alerting=["slack"], **kwargs)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_sgr_soft_limit_alerts_once_per_window():
|
||||
alerting = _alerting()
|
||||
status = _sgr_status(SGRLimitState.SOFT_EXCEEDED)
|
||||
|
||||
await alerting.sgr_limit_alert(status=status)
|
||||
await alerting.sgr_limit_alert(status=status)
|
||||
|
||||
assert len(alerting.sent) == 1
|
||||
level, message = alerting.sent[0]
|
||||
assert level == "Medium"
|
||||
assert "900,000" in message and "1,000,000" in message
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_sgr_hard_limit_alerts_even_after_the_soft_one():
|
||||
alerting = _alerting()
|
||||
|
||||
await alerting.sgr_limit_alert(status=_sgr_status(SGRLimitState.SOFT_EXCEEDED))
|
||||
await alerting.sgr_limit_alert(status=_sgr_status(SGRLimitState.HARD_EXCEEDED))
|
||||
|
||||
assert [level for level, _ in alerting.sent] == ["Medium", "High"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_new_window_alerts_again():
|
||||
alerting = _alerting()
|
||||
|
||||
await alerting.sgr_limit_alert(status=_sgr_status(SGRLimitState.HARD_EXCEEDED, window_start="2026-08-01"))
|
||||
await alerting.sgr_limit_alert(status=_sgr_status(SGRLimitState.HARD_EXCEEDED, window_start="2026-09-01"))
|
||||
|
||||
assert len(alerting.sent) == 2
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_alert_while_under_the_soft_limit():
|
||||
alerting = _alerting()
|
||||
|
||||
await alerting.sgr_limit_alert(status=_sgr_status(SGRLimitState.UNDER))
|
||||
|
||||
assert alerting.sent == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_sgr_alerts_can_be_switched_off_by_alert_type():
|
||||
alerting = _alerting(alert_types=[AlertType.budget_alerts])
|
||||
|
||||
await alerting.sgr_limit_alert(status=_sgr_status(SGRLimitState.HARD_EXCEEDED))
|
||||
|
||||
assert alerting.sent == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_sgr_alerts_are_on_by_default():
|
||||
assert AlertType.sgr_limit_alerts in DEFAULT_ALERT_TYPES
|
||||
|
|
|
|||
170
tests/test_litellm/proxy/db/test_gateway_request_limits.py
Normal file
170
tests/test_litellm/proxy/db/test_gateway_request_limits.py
Normal file
|
|
@ -0,0 +1,170 @@
|
|||
"""
|
||||
Tests for the SGR (successful gateway requests) soft/hard limit: where the
|
||||
allowance is resolved from, how the window is bounded, and what state the
|
||||
window's count lands in.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.proxy.db.gateway_request_limits import (
|
||||
evaluate_sgr_limit,
|
||||
get_sgr_limit_status,
|
||||
resolve_sgr_limit,
|
||||
sgr_window_start,
|
||||
)
|
||||
from litellm.types.proxy.gateway_requests import (
|
||||
SGRLimitConfig,
|
||||
SGRLimitState,
|
||||
SGRLimitWindow,
|
||||
)
|
||||
|
||||
# ── resolving the allowance ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_no_allowance_without_a_license_or_config():
|
||||
assert resolve_sgr_limit(general_settings={}, license_data=None) is None
|
||||
|
||||
|
||||
def test_license_max_sgr_is_the_allowance():
|
||||
config = resolve_sgr_limit(general_settings={}, license_data={"max_sgr": 1_000_000})
|
||||
assert config == SGRLimitConfig(limit=1_000_000, soft_limit=800_000, window=SGRLimitWindow.MONTH)
|
||||
|
||||
|
||||
def test_config_overrides_the_license_so_a_customer_can_alert_earlier():
|
||||
config = resolve_sgr_limit(general_settings={"sgr_limit": 250_000}, license_data={"max_sgr": 1_000_000})
|
||||
assert config is not None
|
||||
assert config.limit == 250_000
|
||||
|
||||
|
||||
def test_soft_limit_percent_moves_the_soft_threshold():
|
||||
config = resolve_sgr_limit(
|
||||
general_settings={"sgr_limit": 1000, "sgr_soft_limit_percent": 0.5},
|
||||
license_data=None,
|
||||
)
|
||||
assert config is not None
|
||||
assert config.soft_limit == 500
|
||||
|
||||
|
||||
def test_window_can_be_the_calendar_year():
|
||||
config = resolve_sgr_limit(general_settings={"sgr_limit": 10, "sgr_limit_window": "year"}, license_data=None)
|
||||
assert config is not None
|
||||
assert config.window is SGRLimitWindow.YEAR
|
||||
|
||||
|
||||
def test_a_license_without_max_sgr_carries_no_allowance():
|
||||
assert resolve_sgr_limit(general_settings={}, license_data={"max_users": 5}) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("max_sgr", [0, -1, "1000", None])
|
||||
def test_a_nonsense_license_value_is_not_an_allowance(max_sgr: object):
|
||||
assert resolve_sgr_limit(general_settings={}, license_data={"max_sgr": max_sgr}) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"general_settings",
|
||||
[
|
||||
{"sgr_limit": 0},
|
||||
{"sgr_limit": -5},
|
||||
{"sgr_limit": "lots"},
|
||||
{"sgr_limit": 100, "sgr_soft_limit_percent": 1.5},
|
||||
{"sgr_limit": 100, "sgr_soft_limit_percent": 0},
|
||||
{"sgr_limit": 100, "sgr_limit_window": "fortnight"},
|
||||
],
|
||||
)
|
||||
def test_invalid_settings_disable_the_limit_instead_of_crashing_the_proxy(general_settings: dict):
|
||||
assert resolve_sgr_limit(general_settings=general_settings, license_data=None) is None
|
||||
|
||||
|
||||
def test_settings_left_unset_fall_back_to_the_license():
|
||||
config = resolve_sgr_limit(
|
||||
general_settings={"sgr_limit": None, "sgr_soft_limit_percent": None, "sgr_limit_window": None},
|
||||
license_data={"max_sgr": 100},
|
||||
)
|
||||
assert config == SGRLimitConfig(limit=100, soft_limit=80, window=SGRLimitWindow.MONTH)
|
||||
|
||||
|
||||
def test_a_tiny_allowance_still_has_a_soft_threshold_of_at_least_one():
|
||||
config = resolve_sgr_limit(general_settings={"sgr_limit": 1}, license_data=None)
|
||||
assert config is not None
|
||||
assert config.soft_limit == 1
|
||||
|
||||
|
||||
# ── the window ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"window, expected",
|
||||
[(SGRLimitWindow.MONTH, "2026-08-01"), (SGRLimitWindow.YEAR, "2026-01-01")],
|
||||
)
|
||||
def test_window_starts_on_the_calendar_boundary(window: SGRLimitWindow, expected: str):
|
||||
assert sgr_window_start(window, datetime(2026, 8, 8, 13, 45, tzinfo=timezone.utc)) == expected
|
||||
|
||||
|
||||
# ── evaluating a count ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"successful_requests, expected",
|
||||
[
|
||||
(0, SGRLimitState.UNDER),
|
||||
(799, SGRLimitState.UNDER),
|
||||
(800, SGRLimitState.SOFT_EXCEEDED),
|
||||
(999, SGRLimitState.SOFT_EXCEEDED),
|
||||
(1000, SGRLimitState.HARD_EXCEEDED),
|
||||
(5000, SGRLimitState.HARD_EXCEEDED),
|
||||
],
|
||||
)
|
||||
def test_thresholds_are_inclusive(successful_requests: int, expected: SGRLimitState):
|
||||
config = SGRLimitConfig(limit=1000, soft_limit=800, window=SGRLimitWindow.MONTH)
|
||||
assert evaluate_sgr_limit(config, successful_requests) is expected
|
||||
|
||||
|
||||
# ── reading the window's count ────────────────────────────────────────────────
|
||||
|
||||
|
||||
class FakeDB:
|
||||
def __init__(self, rows: list[dict]) -> None:
|
||||
self.rows = rows
|
||||
self.queries: list[tuple[str, tuple[object, ...]]] = []
|
||||
|
||||
async def query_raw(self, sql: str, *args: object) -> list[dict]:
|
||||
self.queries.append((sql, args))
|
||||
return self.rows
|
||||
|
||||
|
||||
class FakePrismaClient:
|
||||
def __init__(self, rows: list[dict]) -> None:
|
||||
self.db = FakeDB(rows)
|
||||
|
||||
|
||||
def _status(rows: list[dict], config: SGRLimitConfig):
|
||||
client = FakePrismaClient(rows)
|
||||
status = asyncio.run(
|
||||
get_sgr_limit_status(
|
||||
prisma_client=client,
|
||||
config=config,
|
||||
now=datetime(2026, 8, 8, tzinfo=timezone.utc),
|
||||
)
|
||||
)
|
||||
return status, client
|
||||
|
||||
|
||||
def test_status_counts_only_the_current_window():
|
||||
config = SGRLimitConfig(limit=1000, soft_limit=800, window=SGRLimitWindow.MONTH)
|
||||
status, client = _status([{"successful_requests": 900}], config)
|
||||
|
||||
assert client.db.queries[0][1] == ("2026-08-01",)
|
||||
assert status.successful_requests == 900
|
||||
assert status.window_start == "2026-08-01"
|
||||
assert status.state is SGRLimitState.SOFT_EXCEEDED
|
||||
|
||||
|
||||
def test_status_of_a_deployment_that_has_served_nothing_yet():
|
||||
config = SGRLimitConfig(limit=1000, soft_limit=800, window=SGRLimitWindow.MONTH)
|
||||
status, _ = _status([], config)
|
||||
|
||||
assert status.successful_requests == 0
|
||||
assert status.state is SGRLimitState.UNDER
|
||||
|
|
@ -300,4 +300,41 @@ class TestGatewayDailyActivityRoute:
|
|||
"failed_requests": 3,
|
||||
}
|
||||
],
|
||||
"sgr_limit": None,
|
||||
}
|
||||
|
||||
|
||||
class TestSGRLimitOnTheActivityEndpoint:
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_limit_status_when_the_deployment_has_no_allowance(self):
|
||||
with patch("litellm.proxy.proxy_server.prisma_client", _prisma_returning([])):
|
||||
with patch("litellm.proxy.proxy_server.general_settings", {}):
|
||||
response = await get_gateway_daily_activity(user_api_key_dict=_admin())
|
||||
assert response.sgr_limit is None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_reports_the_windows_count_against_a_configured_allowance(self):
|
||||
client = MagicMock()
|
||||
client.db = MagicMock()
|
||||
client.db.query_raw = AsyncMock(side_effect=[[], [{"successful_requests": 900}]])
|
||||
with patch("litellm.proxy.proxy_server.prisma_client", client):
|
||||
with patch("litellm.proxy.proxy_server.general_settings", {"sgr_limit": 1000}):
|
||||
response = await get_gateway_daily_activity(user_api_key_dict=_admin())
|
||||
|
||||
assert response.sgr_limit is not None
|
||||
assert response.sgr_limit.limit == 1000
|
||||
assert response.sgr_limit.soft_limit == 800
|
||||
assert response.sgr_limit.successful_requests == 900
|
||||
assert response.sgr_limit.state.value == "soft_exceeded"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_failed_limit_read_still_serves_the_activity_numbers(self):
|
||||
client = MagicMock()
|
||||
client.db = MagicMock()
|
||||
client.db.query_raw = AsyncMock(side_effect=[[], RuntimeError("limit query failed")])
|
||||
with patch("litellm.proxy.proxy_server.prisma_client", client):
|
||||
with patch("litellm.proxy.proxy_server.general_settings", {"sgr_limit": 1000}):
|
||||
response = await get_gateway_daily_activity(user_api_key_dict=_admin())
|
||||
|
||||
assert response.sgr_limit is None
|
||||
assert response.total_successful_requests == 0
|
||||
|
|
|
|||
|
|
@ -689,6 +689,55 @@ describe("UsagePage", () => {
|
|||
expect(screen.queryByTestId("gateway-requests-by-endpoint")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("should not show an SGR limit banner when no limit is configured", async () => {
|
||||
renderWithProviders(<UsagePage {...defaultProps} />);
|
||||
|
||||
await waitFor(() => {
|
||||
expect(mockGatewayDailyActivityCall).toHaveBeenCalled();
|
||||
});
|
||||
expect(screen.queryByText(/gateway request limit/i)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("should warn on the usage page once the SGR soft limit is crossed", async () => {
|
||||
mockGatewayDailyActivityCall.mockResolvedValue({
|
||||
...mockGatewayActivity,
|
||||
sgr_limit: {
|
||||
limit: 1000000,
|
||||
soft_limit: 800000,
|
||||
window: "month",
|
||||
window_start: "2026-08-01",
|
||||
successful_requests: 900000,
|
||||
state: "soft_exceeded",
|
||||
},
|
||||
});
|
||||
|
||||
renderWithProviders(<UsagePage {...defaultProps} />);
|
||||
|
||||
const headline = await screen.findByText(/Approaching the gateway request limit/i);
|
||||
expect(headline).toHaveTextContent("900,000 of 1,000,000 successful requests this month");
|
||||
expect(headline.closest("[data-testid='antd-alert']")).toHaveAttribute("data-type", "warning");
|
||||
});
|
||||
|
||||
it("should escalate the banner once the SGR limit itself is reached", async () => {
|
||||
mockGatewayDailyActivityCall.mockResolvedValue({
|
||||
...mockGatewayActivity,
|
||||
sgr_limit: {
|
||||
limit: 1000000,
|
||||
soft_limit: 800000,
|
||||
window: "month",
|
||||
window_start: "2026-08-01",
|
||||
successful_requests: 1100000,
|
||||
state: "hard_exceeded",
|
||||
},
|
||||
});
|
||||
|
||||
renderWithProviders(<UsagePage {...defaultProps} />);
|
||||
|
||||
const headline = await screen.findByText(/Gateway request limit reached/i);
|
||||
expect(headline).toHaveTextContent("1,100,000 of 1,000,000 successful requests this month");
|
||||
expect(headline.closest("[data-testid='antd-alert']")).toHaveAttribute("data-type", "error");
|
||||
});
|
||||
|
||||
it("should display usage metrics and charts", async () => {
|
||||
renderWithProviders(<UsagePage {...defaultProps} />);
|
||||
|
||||
|
|
|
|||
|
|
@ -58,6 +58,7 @@ import {
|
|||
fetchedRangeKey,
|
||||
selectForRange,
|
||||
selectGatewayActivity,
|
||||
sgrLimitBanner,
|
||||
topGatewayRoutes,
|
||||
type FetchedForRange,
|
||||
type FetchedGatewayActivity,
|
||||
|
|
@ -494,6 +495,7 @@ const UsagePage: React.FC<UsagePageProps> = ({ teams, organizations }) => {
|
|||
[userSpendData.results],
|
||||
);
|
||||
const gatewayRequestsByRoute = useMemo(() => topGatewayRoutes(gatewayActivity), [gatewayActivity]);
|
||||
const sgrBanner = useMemo(() => sgrLimitBanner(gatewayActivity), [gatewayActivity]);
|
||||
const modelMetrics = useMemo(
|
||||
() => processActivityData(userSpendData, modelViewType === "groups" ? "model_groups" : "models", teams),
|
||||
[userSpendData, modelViewType, teams],
|
||||
|
|
@ -555,6 +557,17 @@ const UsagePage: React.FC<UsagePageProps> = ({ teams, organizations }) => {
|
|||
}
|
||||
/>
|
||||
)}
|
||||
{sgrBanner && (
|
||||
<Alert
|
||||
banner
|
||||
showIcon
|
||||
data-testid="sgr-limit-banner"
|
||||
type={sgrBanner.severity}
|
||||
className="mb-2"
|
||||
message={sgrBanner.headline}
|
||||
description={sgrBanner.detail}
|
||||
/>
|
||||
)}
|
||||
{/* Your Usage / Global Usage Panel */}
|
||||
{(usageView === "global" || usageView === "my-usage") && (
|
||||
<>
|
||||
|
|
|
|||
|
|
@ -4,8 +4,10 @@ import {
|
|||
fetchedRangeKey,
|
||||
selectForRange,
|
||||
selectGatewayActivity,
|
||||
sgrLimitBanner,
|
||||
topGatewayRoutes,
|
||||
type GatewayActivity,
|
||||
type SGRLimit,
|
||||
} from "./gatewayActivity";
|
||||
|
||||
const activity = (total: number): GatewayActivity => ({
|
||||
|
|
@ -106,3 +108,39 @@ describe("topGatewayRoutes", () => {
|
|||
expect(topGatewayRoutes(null)).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("sgrLimitBanner", () => {
|
||||
const withLimit = (sgrLimit: SGRLimit | null): GatewayActivity => ({ ...activity(0), sgr_limit: sgrLimit });
|
||||
|
||||
const limit: SGRLimit = {
|
||||
limit: 1_000_000,
|
||||
soft_limit: 800_000,
|
||||
window: "month",
|
||||
window_start: "2026-08-01",
|
||||
successful_requests: 900_000,
|
||||
state: "soft_exceeded",
|
||||
};
|
||||
|
||||
it("says nothing when the deployment has no allowance, or is under it", () => {
|
||||
expect(sgrLimitBanner(null)).toBeNull();
|
||||
expect(sgrLimitBanner(activity(0))).toBeNull();
|
||||
expect(sgrLimitBanner(withLimit(null))).toBeNull();
|
||||
expect(sgrLimitBanner(withLimit({ ...limit, successful_requests: 10, state: "under" }))).toBeNull();
|
||||
});
|
||||
|
||||
it("warns once past the soft threshold, naming the count, the limit and the window", () => {
|
||||
const banner = sgrLimitBanner(withLimit(limit));
|
||||
expect(banner?.severity).toEqual("warning");
|
||||
expect(banner?.headline).toContain("900,000 of 1,000,000");
|
||||
expect(banner?.headline).toContain("this month");
|
||||
expect(banner?.detail).toContain("80%");
|
||||
expect(banner?.detail).toContain("2026-08-01");
|
||||
});
|
||||
|
||||
it("escalates to an error once the limit itself is reached, and says traffic is unaffected", () => {
|
||||
const banner = sgrLimitBanner(withLimit({ ...limit, successful_requests: 1_100_000, state: "hard_exceeded" }));
|
||||
expect(banner?.severity).toEqual("error");
|
||||
expect(banner?.headline).toContain("1,100,000 of 1,000,000");
|
||||
expect(banner?.detail).toContain("still being served");
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -10,13 +10,62 @@
|
|||
|
||||
export const GATEWAY_TOP_ROUTES = 15;
|
||||
|
||||
export type SGRLimitState = "under" | "soft_exceeded" | "hard_exceeded";
|
||||
|
||||
/**
|
||||
* Standing against the configured SGR allowance, for the limit's own window.
|
||||
*
|
||||
* `successful_requests` here is not the total above: that one covers the
|
||||
* selected date range, this one covers the window the limit is counted over.
|
||||
*/
|
||||
export interface SGRLimit {
|
||||
limit: number;
|
||||
soft_limit: number;
|
||||
window: "month" | "year";
|
||||
window_start: string;
|
||||
successful_requests: number;
|
||||
state: SGRLimitState;
|
||||
}
|
||||
|
||||
export interface GatewayActivity {
|
||||
total_successful_requests: number;
|
||||
total_failed_requests: number;
|
||||
by_date: { date: string; successful_requests: number; failed_requests: number }[];
|
||||
by_route: { category: string; route: string; successful_requests: number; failed_requests: number }[];
|
||||
/** Absent on a deployment with no allowance configured, and on older proxies. */
|
||||
sgr_limit?: SGRLimit | null;
|
||||
}
|
||||
|
||||
export interface SGRLimitBanner {
|
||||
severity: "warning" | "error";
|
||||
headline: string;
|
||||
detail: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* The banner to show for the SGR allowance, or null when there is nothing to say.
|
||||
*
|
||||
* Nothing to say covers three cases: no allowance configured, an allowance the
|
||||
* deployment is still under, and a proxy too old to report one.
|
||||
*/
|
||||
export const sgrLimitBanner = (activity: GatewayActivity | null): SGRLimitBanner | null => {
|
||||
const sgrLimit = activity?.sgr_limit;
|
||||
if (sgrLimit == null || sgrLimit.state === "under") return null;
|
||||
const used = `${sgrLimit.successful_requests.toLocaleString()} of ${sgrLimit.limit.toLocaleString()}`;
|
||||
const percent = Math.floor((100 * sgrLimit.soft_limit) / sgrLimit.limit);
|
||||
return sgrLimit.state === "hard_exceeded"
|
||||
? {
|
||||
severity: "error",
|
||||
headline: `Gateway request limit reached: ${used} successful requests this ${sgrLimit.window}`,
|
||||
detail: `Counted since ${sgrLimit.window_start} (UTC). Requests are still being served, this is an alert only.`,
|
||||
}
|
||||
: {
|
||||
severity: "warning",
|
||||
headline: `Approaching the gateway request limit: ${used} successful requests this ${sgrLimit.window}`,
|
||||
detail: `Past ${percent}% of the limit, counted since ${sgrLimit.window_start} (UTC).`,
|
||||
};
|
||||
};
|
||||
|
||||
/** A fetched result carrying the range key it was fetched for. */
|
||||
export interface FetchedForRange<T> {
|
||||
rangeKey: string;
|
||||
|
|
|
|||
53
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
53
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -21237,7 +21237,7 @@ export interface components {
|
|||
* @description Enum for alert types and management event types
|
||||
* @enum {string}
|
||||
*/
|
||||
AlertType: "llm_exceptions" | "llm_too_slow" | "llm_requests_hanging" | "budget_alerts" | "spend_reports" | "failed_tracking_spend" | "db_exceptions" | "daily_reports" | "cooldown_deployment" | "new_model_added" | "outage_alerts" | "region_outage_alerts" | "fallback_reports" | "new_virtual_key_created" | "virtual_key_updated" | "virtual_key_deleted" | "new_team_created" | "team_updated" | "team_deleted" | "new_internal_user_created" | "internal_user_updated" | "internal_user_deleted";
|
||||
AlertType: "llm_exceptions" | "llm_too_slow" | "llm_requests_hanging" | "budget_alerts" | "spend_reports" | "failed_tracking_spend" | "sgr_limit_alerts" | "db_exceptions" | "daily_reports" | "cooldown_deployment" | "new_model_added" | "outage_alerts" | "region_outage_alerts" | "fallback_reports" | "new_virtual_key_created" | "virtual_key_updated" | "virtual_key_deleted" | "new_team_created" | "team_updated" | "team_deleted" | "new_internal_user_created" | "internal_user_updated" | "internal_user_deleted";
|
||||
/** AllowedVectorStoreIndexItem */
|
||||
AllowedVectorStoreIndexItem: {
|
||||
/** Index Name */
|
||||
|
|
@ -21391,6 +21391,13 @@ export interface components {
|
|||
* @description What the routed traffic actually cost
|
||||
*/
|
||||
spend: number;
|
||||
/**
|
||||
* Tier Turns
|
||||
* @description Turns per tier, keyed by the tier name the routing decision recorded at request time (never re-derived at read time, since the tier-to-model mapping is mutable config). Tier names are scoped to this group's router_type and are not comparable across types: a complexity router reports 'simple'/'medium'/'complex'/'reasoning', a quality router reports its numeric quality tier, and an adaptive router records no tier at all. Turns no tier served (the classifier fell back to default_model) are absent rather than pooled under a sentinel key, so the values may sum to less than turns
|
||||
*/
|
||||
tier_turns?: {
|
||||
[key: string]: number;
|
||||
};
|
||||
/** Turns */
|
||||
turns: number;
|
||||
};
|
||||
|
|
@ -23621,6 +23628,21 @@ export interface components {
|
|||
* @description When set to True, rejects requests that contain client-side 'metadata.tags' to prevent users from influencing budgets by sending different tags. Tags can only be inherited from the API key metadata.
|
||||
*/
|
||||
reject_clientside_metadata_tags?: boolean | null;
|
||||
/**
|
||||
* Sgr Limit
|
||||
* @description Limit on successful gateway requests (SGR) per window. Crossing it alerts on the admin UI and over any configured alerting integration, it does not reject traffic. Takes precedence over an enterprise license's `max_sgr`
|
||||
*/
|
||||
sgr_limit?: number | null;
|
||||
/**
|
||||
* Sgr Limit Window
|
||||
* @description Period `sgr_limit` is counted over, calendar aligned in UTC. Defaults to `month`
|
||||
*/
|
||||
sgr_limit_window?: ("month" | "year") | null;
|
||||
/**
|
||||
* Sgr Soft Limit Percent
|
||||
* @description Fraction of `sgr_limit` at which the soft alert fires. Defaults to 0.8
|
||||
*/
|
||||
sgr_soft_limit_percent?: number | null;
|
||||
/**
|
||||
* Store Model In Db
|
||||
* @description If True, models and config are stored in and loaded from the database. Default is False.
|
||||
|
|
@ -24860,6 +24882,8 @@ export interface components {
|
|||
* @default []
|
||||
*/
|
||||
by_route: components["schemas"]["GatewayRequestBreakdownEntry"][];
|
||||
/** @description Standing against the configured SGR allowance, or null when no allowance is configured. The totals above cover the requested date range, this covers the limit window, so the two counts are not the same number */
|
||||
sgr_limit?: components["schemas"]["SGRLimitStatus"] | null;
|
||||
/**
|
||||
* Total Failed Requests
|
||||
* @default 0
|
||||
|
|
@ -32227,6 +32251,33 @@ export interface components {
|
|||
/** Middlename */
|
||||
middleName?: string | null;
|
||||
};
|
||||
/**
|
||||
* SGRLimitState
|
||||
* @enum {string}
|
||||
*/
|
||||
SGRLimitState: "soft_exceeded" | "hard_exceeded" | "under";
|
||||
/**
|
||||
* SGRLimitStatus
|
||||
* @description Where the deployment sits against its SGR allowance for the current window.
|
||||
*/
|
||||
SGRLimitStatus: {
|
||||
/** Limit */
|
||||
limit: number;
|
||||
/** Soft Limit */
|
||||
soft_limit: number;
|
||||
state: components["schemas"]["SGRLimitState"];
|
||||
/** Successful Requests */
|
||||
successful_requests: number;
|
||||
window: components["schemas"]["SGRLimitWindow"];
|
||||
/** Window Start */
|
||||
window_start: string;
|
||||
};
|
||||
/**
|
||||
* SGRLimitWindow
|
||||
* @description Period an SGR allowance is counted over. Calendar aligned and UTC.
|
||||
* @enum {string}
|
||||
*/
|
||||
SGRLimitWindow: "month" | "year";
|
||||
/**
|
||||
* SSOConfig
|
||||
* @description Configuration for SSO environment variables and settings
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue