mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
feat(proxy): add project-level ITPM and OTPM quotas
Add model_itpm_limit and model_otpm_limit to project create and update requests, storing both quota maps in project metadata without a database migration Reserve input and output tokens independently before provider dispatch, expose separate project rate-limit headers, and reconcile counters across successful calls, failures, retries, fallbacks, streaming, caching, and cancellation Harden token estimation for pre-tokenized embeddings, multimodal inputs, Responses API requests, native Gemini requests, multiple candidates, and conflicting output-cap aliases Reject negative output caps, preserve conservative reservations when usage is missing or zero, bind reconciliation and refunds to the reservation window, prevent double refunds or negative counters, update generated API types, and add regression coverage
This commit is contained in:
parent
4fcaf7d736
commit
9dbe61aa6d
8 changed files with 4095 additions and 90 deletions
|
|
@ -1,7 +1,7 @@
|
|||
import enum
|
||||
import json
|
||||
import os
|
||||
from collections.abc import Callable
|
||||
from collections.abc import Callable, Mapping
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal, Union
|
||||
|
||||
|
|
@ -2903,6 +2903,8 @@ class NewProjectRequest(LiteLLM_BudgetTable):
|
|||
models: list[str] = []
|
||||
model_rpm_limit: dict | None = None
|
||||
model_tpm_limit: dict | None = None
|
||||
model_itpm_limit: Mapping[str, int] | None = None
|
||||
model_otpm_limit: Mapping[str, int] | None = None
|
||||
blocked: bool = False
|
||||
object_permission: LiteLLM_ObjectPermissionBase | None = None
|
||||
|
||||
|
|
@ -2935,6 +2937,8 @@ class UpdateProjectRequest(LiteLLM_BudgetTable):
|
|||
models: list[str] | None = None
|
||||
model_rpm_limit: dict | None = None
|
||||
model_tpm_limit: dict | None = None
|
||||
model_itpm_limit: Mapping[str, int] | None = None
|
||||
model_otpm_limit: Mapping[str, int] | None = None
|
||||
blocked: bool | None = None
|
||||
budget_id: str | None = None
|
||||
object_permission: LiteLLM_ObjectPermissionBase | None = None
|
||||
|
|
@ -4072,6 +4076,8 @@ class PassThroughEndpointLoggingTypedDict(TypedDict):
|
|||
LiteLLM_ManagementEndpoint_MetadataFields: Final = [
|
||||
"model_rpm_limit",
|
||||
"model_tpm_limit",
|
||||
"model_itpm_limit",
|
||||
"model_otpm_limit",
|
||||
"mcp_rpm_limit",
|
||||
"tag_rpm_limit",
|
||||
"rpm_limit_type",
|
||||
|
|
|
|||
|
|
@ -994,7 +994,7 @@ def get_key_model_tpm_limit(
|
|||
def get_model_rate_limit_from_metadata(
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
metadata_accessor_key: Literal["team_metadata", "organization_metadata", "project_metadata"],
|
||||
rate_limit_key: Literal["model_rpm_limit", "model_tpm_limit"],
|
||||
rate_limit_key: Literal["model_rpm_limit", "model_tpm_limit", "model_itpm_limit", "model_otpm_limit"],
|
||||
) -> dict[str, int] | None:
|
||||
if getattr(user_api_key_dict, metadata_accessor_key):
|
||||
return getattr(user_api_key_dict, metadata_accessor_key).get(rate_limit_key)
|
||||
|
|
|
|||
|
|
@ -454,7 +454,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
parent_otel_span=user_api_key_dict.parent_otel_span,
|
||||
)
|
||||
|
||||
verbose_proxy_logger.debug("Atomic check+increment response: %s", json.dumps(atomic_response, indent=2))
|
||||
verbose_proxy_logger.debug(
|
||||
"Atomic check+increment response: %s", json.dumps(atomic_response, indent=2, default=list)
|
||||
)
|
||||
|
||||
if atomic_response["overall_code"] == "OVER_LIMIT":
|
||||
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(model)
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -3114,6 +3114,136 @@ async def test_project_model_rate_limits_not_triggered_for_other_model_v3():
|
|||
), f"model_per_project should not be added for unrelated model, got: {descriptor_keys}"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_project_model_itpm_otpm_limits_enforced_v3():
|
||||
"""
|
||||
Project-level model_itpm_limit/model_otpm_limit must produce distinct
|
||||
Bedrock Mantle-style input and output token descriptors.
|
||||
"""
|
||||
_api_key = hash_token("sk-project-io-test")
|
||||
local_cache = DualCache()
|
||||
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
|
||||
internal_usage_cache=InternalUsageCache(local_cache)
|
||||
)
|
||||
|
||||
captured_descriptors = []
|
||||
|
||||
async def mock_should_rate_limit(descriptors, **kwargs):
|
||||
captured_descriptors.extend(descriptors)
|
||||
return {"overall_code": "OK", "statuses": []}
|
||||
|
||||
parallel_request_handler.should_rate_limit = mock_should_rate_limit
|
||||
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key=_api_key,
|
||||
project_id="proj-mantle",
|
||||
project_metadata={
|
||||
"model_itpm_limit": {"bedrock_mantle/claude-opus": 20000000},
|
||||
"model_otpm_limit": {"bedrock_mantle/claude-opus": 4000000},
|
||||
},
|
||||
)
|
||||
|
||||
await parallel_request_handler.async_pre_call_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
cache=local_cache,
|
||||
data={"model": "bedrock_mantle/claude-opus"},
|
||||
call_type="",
|
||||
)
|
||||
|
||||
descriptor_keys = [d["key"] for d in captured_descriptors]
|
||||
assert "model_per_project_itpm" in descriptor_keys
|
||||
assert "model_per_project_otpm" in descriptor_keys
|
||||
assert "model_per_project" not in descriptor_keys
|
||||
|
||||
itpm_descriptor = next(
|
||||
d for d in captured_descriptors if d["key"] == "model_per_project_itpm"
|
||||
)
|
||||
otpm_descriptor = next(
|
||||
d for d in captured_descriptors if d["key"] == "model_per_project_otpm"
|
||||
)
|
||||
assert itpm_descriptor["value"] == "proj-mantle:bedrock_mantle/claude-opus"
|
||||
assert itpm_descriptor["rate_limit"]["tokens_per_unit"] == 20000000
|
||||
assert otpm_descriptor["value"] == "proj-mantle:bedrock_mantle/claude-opus"
|
||||
assert otpm_descriptor["rate_limit"]["tokens_per_unit"] == 4000000
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_project_model_itpm_otpm_limits_not_triggered_for_other_model_v3():
|
||||
"""Split project limits must not apply to an unrelated model."""
|
||||
_api_key = hash_token("sk-project-io-test-2")
|
||||
local_cache = DualCache()
|
||||
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
|
||||
internal_usage_cache=InternalUsageCache(local_cache)
|
||||
)
|
||||
|
||||
captured_descriptors = []
|
||||
|
||||
async def mock_should_rate_limit(descriptors, **kwargs):
|
||||
captured_descriptors.extend(descriptors)
|
||||
return {"overall_code": "OK", "statuses": []}
|
||||
|
||||
parallel_request_handler.should_rate_limit = mock_should_rate_limit
|
||||
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key=_api_key,
|
||||
project_id="proj-mantle",
|
||||
project_metadata={
|
||||
"model_itpm_limit": {"bedrock_mantle/claude-opus": 20000000},
|
||||
},
|
||||
)
|
||||
|
||||
await parallel_request_handler.async_pre_call_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
cache=local_cache,
|
||||
data={"model": "gpt-4"},
|
||||
call_type="",
|
||||
)
|
||||
|
||||
descriptor_keys = [d["key"] for d in captured_descriptors]
|
||||
assert "model_per_project_itpm" not in descriptor_keys
|
||||
assert "model_per_project_otpm" not in descriptor_keys
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_project_model_itpm_and_tpm_limits_coexist_v3():
|
||||
"""Combined project TPM and split ITPM/OTPM limits are enforced together."""
|
||||
_api_key = hash_token("sk-project-io-test-3")
|
||||
local_cache = DualCache()
|
||||
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
|
||||
internal_usage_cache=InternalUsageCache(local_cache)
|
||||
)
|
||||
|
||||
captured_descriptors = []
|
||||
|
||||
async def mock_should_rate_limit(descriptors, **kwargs):
|
||||
captured_descriptors.extend(descriptors)
|
||||
return {"overall_code": "OK", "statuses": []}
|
||||
|
||||
parallel_request_handler.should_rate_limit = mock_should_rate_limit
|
||||
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key=_api_key,
|
||||
project_id="proj-mantle",
|
||||
project_metadata={
|
||||
"model_tpm_limit": {"bedrock_mantle/claude-opus": 1000},
|
||||
"model_itpm_limit": {"bedrock_mantle/claude-opus": 20000000},
|
||||
"model_otpm_limit": {"bedrock_mantle/claude-opus": 4000000},
|
||||
},
|
||||
)
|
||||
|
||||
await parallel_request_handler.async_pre_call_hook(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
cache=local_cache,
|
||||
data={"model": "bedrock_mantle/claude-opus"},
|
||||
call_type="",
|
||||
)
|
||||
|
||||
descriptor_keys = [d["key"] for d in captured_descriptors]
|
||||
assert "model_per_project" in descriptor_keys
|
||||
assert "model_per_project_itpm" in descriptor_keys
|
||||
assert "model_per_project_otpm" in descriptor_keys
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_pre_call_hook_keeps_internal_stash_out_of_request_body():
|
||||
"""Regression for #27001 / #35197: the limiter's per-request bookkeeping
|
||||
|
|
@ -3190,7 +3320,7 @@ async def test_responses_route_body_untouched_by_pre_call_hook(caller_metadata):
|
|||
_api_key = hash_token("sk-responses-regression")
|
||||
user_api_key_dict = UserAPIKeyAuth(
|
||||
api_key=_api_key,
|
||||
tpm_limit=1000,
|
||||
tpm_limit=100000,
|
||||
rpm_limit=5,
|
||||
)
|
||||
local_cache = DualCache()
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -159,3 +159,21 @@ def test_update_key_request_requires_key_or_key_alias():
|
|||
by_alias = UpdateKeyRequest(key_alias="my-alias")
|
||||
assert by_alias.key is None
|
||||
assert by_alias.key_alias == "my-alias"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("request_type", ["new", "update"])
|
||||
def test_project_io_token_limits_are_stored_in_metadata(request_type):
|
||||
from litellm.proxy._types import NewProjectRequest, UpdateProjectRequest
|
||||
|
||||
limits = {
|
||||
"model_itpm_limit": {"bedrock_mantle/openai.gpt-oss-120b": 20_000_000},
|
||||
"model_otpm_limit": {"bedrock_mantle/openai.gpt-oss-120b": 4_000_000},
|
||||
}
|
||||
request = (
|
||||
NewProjectRequest(team_id="team-1", **limits)
|
||||
if request_type == "new"
|
||||
else UpdateProjectRequest(project_id="project-1", **limits)
|
||||
)
|
||||
|
||||
assert request.metadata == limits
|
||||
assert request.model_dump(exclude_none=True)["metadata"] == limits
|
||||
|
|
|
|||
16
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
16
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -28772,10 +28772,18 @@ export interface components {
|
|||
metadata?: {
|
||||
[key: string]: unknown;
|
||||
} | null;
|
||||
/** Model Itpm Limit */
|
||||
model_itpm_limit?: {
|
||||
[key: string]: number;
|
||||
} | null;
|
||||
/** Model Max Budget */
|
||||
model_max_budget?: {
|
||||
[key: string]: unknown;
|
||||
} | null;
|
||||
/** Model Otpm Limit */
|
||||
model_otpm_limit?: {
|
||||
[key: string]: number;
|
||||
} | null;
|
||||
/** Model Rpm Limit */
|
||||
model_rpm_limit?: {
|
||||
[key: string]: unknown;
|
||||
|
|
@ -33623,10 +33631,18 @@ export interface components {
|
|||
metadata?: {
|
||||
[key: string]: unknown;
|
||||
} | null;
|
||||
/** Model Itpm Limit */
|
||||
model_itpm_limit?: {
|
||||
[key: string]: number;
|
||||
} | null;
|
||||
/** Model Max Budget */
|
||||
model_max_budget?: {
|
||||
[key: string]: unknown;
|
||||
} | null;
|
||||
/** Model Otpm Limit */
|
||||
model_otpm_limit?: {
|
||||
[key: string]: number;
|
||||
} | null;
|
||||
/** Model Rpm Limit */
|
||||
model_rpm_limit?: {
|
||||
[key: string]: unknown;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue