feat(proxy): add project-level ITPM and OTPM quotas

Add model_itpm_limit and model_otpm_limit to project create and update requests, storing both quota maps in project metadata without a database migration

Reserve input and output tokens independently before provider dispatch, expose separate project rate-limit headers, and reconcile counters across successful calls, failures, retries, fallbacks, streaming, caching, and cancellation

Harden token estimation for pre-tokenized embeddings, multimodal inputs, Responses API requests, native Gemini requests, multiple candidates, and conflicting output-cap aliases

Reject negative output caps, preserve conservative reservations when usage is missing or zero, bind reconciliation and refunds to the reservation window, prevent double refunds or negative counters, update generated API types, and add regression coverage
This commit is contained in:
Shivi Jain 2026-07-31 16:49:07 +05:30
parent 4fcaf7d736
commit 9dbe61aa6d
8 changed files with 4095 additions and 90 deletions

View file

@ -1,7 +1,7 @@
import enum
import json
import os
from collections.abc import Callable
from collections.abc import Callable, Mapping
from datetime import datetime
from typing import TYPE_CHECKING, Any, Final, Literal, Union
@ -2903,6 +2903,8 @@ class NewProjectRequest(LiteLLM_BudgetTable):
models: list[str] = []
model_rpm_limit: dict | None = None
model_tpm_limit: dict | None = None
model_itpm_limit: Mapping[str, int] | None = None
model_otpm_limit: Mapping[str, int] | None = None
blocked: bool = False
object_permission: LiteLLM_ObjectPermissionBase | None = None
@ -2935,6 +2937,8 @@ class UpdateProjectRequest(LiteLLM_BudgetTable):
models: list[str] | None = None
model_rpm_limit: dict | None = None
model_tpm_limit: dict | None = None
model_itpm_limit: Mapping[str, int] | None = None
model_otpm_limit: Mapping[str, int] | None = None
blocked: bool | None = None
budget_id: str | None = None
object_permission: LiteLLM_ObjectPermissionBase | None = None
@ -4072,6 +4076,8 @@ class PassThroughEndpointLoggingTypedDict(TypedDict):
LiteLLM_ManagementEndpoint_MetadataFields: Final = [
"model_rpm_limit",
"model_tpm_limit",
"model_itpm_limit",
"model_otpm_limit",
"mcp_rpm_limit",
"tag_rpm_limit",
"rpm_limit_type",

View file

@ -994,7 +994,7 @@ def get_key_model_tpm_limit(
def get_model_rate_limit_from_metadata(
user_api_key_dict: UserAPIKeyAuth,
metadata_accessor_key: Literal["team_metadata", "organization_metadata", "project_metadata"],
rate_limit_key: Literal["model_rpm_limit", "model_tpm_limit"],
rate_limit_key: Literal["model_rpm_limit", "model_tpm_limit", "model_itpm_limit", "model_otpm_limit"],
) -> dict[str, int] | None:
if getattr(user_api_key_dict, metadata_accessor_key):
return getattr(user_api_key_dict, metadata_accessor_key).get(rate_limit_key)

View file

@ -454,7 +454,9 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
parent_otel_span=user_api_key_dict.parent_otel_span,
)
verbose_proxy_logger.debug("Atomic check+increment response: %s", json.dumps(atomic_response, indent=2))
verbose_proxy_logger.debug(
"Atomic check+increment response: %s", json.dumps(atomic_response, indent=2, default=list)
)
if atomic_response["overall_code"] == "OVER_LIMIT":
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(model)

File diff suppressed because it is too large Load diff

View file

@ -3114,6 +3114,136 @@ async def test_project_model_rate_limits_not_triggered_for_other_model_v3():
), f"model_per_project should not be added for unrelated model, got: {descriptor_keys}"
@pytest.mark.asyncio
async def test_project_model_itpm_otpm_limits_enforced_v3():
"""
Project-level model_itpm_limit/model_otpm_limit must produce distinct
Bedrock Mantle-style input and output token descriptors.
"""
_api_key = hash_token("sk-project-io-test")
local_cache = DualCache()
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
internal_usage_cache=InternalUsageCache(local_cache)
)
captured_descriptors = []
async def mock_should_rate_limit(descriptors, **kwargs):
captured_descriptors.extend(descriptors)
return {"overall_code": "OK", "statuses": []}
parallel_request_handler.should_rate_limit = mock_should_rate_limit
user_api_key_dict = UserAPIKeyAuth(
api_key=_api_key,
project_id="proj-mantle",
project_metadata={
"model_itpm_limit": {"bedrock_mantle/claude-opus": 20000000},
"model_otpm_limit": {"bedrock_mantle/claude-opus": 4000000},
},
)
await parallel_request_handler.async_pre_call_hook(
user_api_key_dict=user_api_key_dict,
cache=local_cache,
data={"model": "bedrock_mantle/claude-opus"},
call_type="",
)
descriptor_keys = [d["key"] for d in captured_descriptors]
assert "model_per_project_itpm" in descriptor_keys
assert "model_per_project_otpm" in descriptor_keys
assert "model_per_project" not in descriptor_keys
itpm_descriptor = next(
d for d in captured_descriptors if d["key"] == "model_per_project_itpm"
)
otpm_descriptor = next(
d for d in captured_descriptors if d["key"] == "model_per_project_otpm"
)
assert itpm_descriptor["value"] == "proj-mantle:bedrock_mantle/claude-opus"
assert itpm_descriptor["rate_limit"]["tokens_per_unit"] == 20000000
assert otpm_descriptor["value"] == "proj-mantle:bedrock_mantle/claude-opus"
assert otpm_descriptor["rate_limit"]["tokens_per_unit"] == 4000000
@pytest.mark.asyncio
async def test_project_model_itpm_otpm_limits_not_triggered_for_other_model_v3():
"""Split project limits must not apply to an unrelated model."""
_api_key = hash_token("sk-project-io-test-2")
local_cache = DualCache()
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
internal_usage_cache=InternalUsageCache(local_cache)
)
captured_descriptors = []
async def mock_should_rate_limit(descriptors, **kwargs):
captured_descriptors.extend(descriptors)
return {"overall_code": "OK", "statuses": []}
parallel_request_handler.should_rate_limit = mock_should_rate_limit
user_api_key_dict = UserAPIKeyAuth(
api_key=_api_key,
project_id="proj-mantle",
project_metadata={
"model_itpm_limit": {"bedrock_mantle/claude-opus": 20000000},
},
)
await parallel_request_handler.async_pre_call_hook(
user_api_key_dict=user_api_key_dict,
cache=local_cache,
data={"model": "gpt-4"},
call_type="",
)
descriptor_keys = [d["key"] for d in captured_descriptors]
assert "model_per_project_itpm" not in descriptor_keys
assert "model_per_project_otpm" not in descriptor_keys
@pytest.mark.asyncio
async def test_project_model_itpm_and_tpm_limits_coexist_v3():
"""Combined project TPM and split ITPM/OTPM limits are enforced together."""
_api_key = hash_token("sk-project-io-test-3")
local_cache = DualCache()
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
internal_usage_cache=InternalUsageCache(local_cache)
)
captured_descriptors = []
async def mock_should_rate_limit(descriptors, **kwargs):
captured_descriptors.extend(descriptors)
return {"overall_code": "OK", "statuses": []}
parallel_request_handler.should_rate_limit = mock_should_rate_limit
user_api_key_dict = UserAPIKeyAuth(
api_key=_api_key,
project_id="proj-mantle",
project_metadata={
"model_tpm_limit": {"bedrock_mantle/claude-opus": 1000},
"model_itpm_limit": {"bedrock_mantle/claude-opus": 20000000},
"model_otpm_limit": {"bedrock_mantle/claude-opus": 4000000},
},
)
await parallel_request_handler.async_pre_call_hook(
user_api_key_dict=user_api_key_dict,
cache=local_cache,
data={"model": "bedrock_mantle/claude-opus"},
call_type="",
)
descriptor_keys = [d["key"] for d in captured_descriptors]
assert "model_per_project" in descriptor_keys
assert "model_per_project_itpm" in descriptor_keys
assert "model_per_project_otpm" in descriptor_keys
@pytest.mark.asyncio
async def test_pre_call_hook_keeps_internal_stash_out_of_request_body():
"""Regression for #27001 / #35197: the limiter's per-request bookkeeping
@ -3190,7 +3320,7 @@ async def test_responses_route_body_untouched_by_pre_call_hook(caller_metadata):
_api_key = hash_token("sk-responses-regression")
user_api_key_dict = UserAPIKeyAuth(
api_key=_api_key,
tpm_limit=1000,
tpm_limit=100000,
rpm_limit=5,
)
local_cache = DualCache()

File diff suppressed because it is too large Load diff

View file

@ -159,3 +159,21 @@ def test_update_key_request_requires_key_or_key_alias():
by_alias = UpdateKeyRequest(key_alias="my-alias")
assert by_alias.key is None
assert by_alias.key_alias == "my-alias"
@pytest.mark.parametrize("request_type", ["new", "update"])
def test_project_io_token_limits_are_stored_in_metadata(request_type):
from litellm.proxy._types import NewProjectRequest, UpdateProjectRequest
limits = {
"model_itpm_limit": {"bedrock_mantle/openai.gpt-oss-120b": 20_000_000},
"model_otpm_limit": {"bedrock_mantle/openai.gpt-oss-120b": 4_000_000},
}
request = (
NewProjectRequest(team_id="team-1", **limits)
if request_type == "new"
else UpdateProjectRequest(project_id="project-1", **limits)
)
assert request.metadata == limits
assert request.model_dump(exclude_none=True)["metadata"] == limits

View file

@ -28772,10 +28772,18 @@ export interface components {
metadata?: {
[key: string]: unknown;
} | null;
/** Model Itpm Limit */
model_itpm_limit?: {
[key: string]: number;
} | null;
/** Model Max Budget */
model_max_budget?: {
[key: string]: unknown;
} | null;
/** Model Otpm Limit */
model_otpm_limit?: {
[key: string]: number;
} | null;
/** Model Rpm Limit */
model_rpm_limit?: {
[key: string]: unknown;
@ -33623,10 +33631,18 @@ export interface components {
metadata?: {
[key: string]: unknown;
} | null;
/** Model Itpm Limit */
model_itpm_limit?: {
[key: string]: number;
} | null;
/** Model Max Budget */
model_max_budget?: {
[key: string]: unknown;
} | null;
/** Model Otpm Limit */
model_otpm_limit?: {
[key: string]: number;
} | null;
/** Model Rpm Limit */
model_rpm_limit?: {
[key: string]: unknown;