fix: req changes + UI fixed

This commit is contained in:
Harshit28j 2026-03-07 09:03:36 +05:30
parent 86b5e2d873
commit c39432cc04
3 changed files with 54 additions and 9 deletions

View file

@ -129,7 +129,11 @@ class _PROXY_MaxBudgetPerSessionHandler(CustomLogger):
"""
try:
litellm_params = kwargs.get("litellm_params") or {}
metadata = litellm_params.get("metadata") or {}
metadata = (
litellm_params.get("metadata")
or litellm_params.get("litellm_metadata")
or {}
)
session_id = metadata.get("session_id")
if session_id is None:
return
@ -208,9 +212,7 @@ class _PROXY_MaxBudgetPerSessionHandler(CustomLogger):
async def _get_current_spend(self, cache_key: str) -> float:
"""Read current accumulated spend for a session."""
if (
self.internal_usage_cache.dual_cache.redis_cache is not None
):
if self.internal_usage_cache.dual_cache.redis_cache is not None:
try:
result = await self.internal_usage_cache.dual_cache.redis_cache.async_get_cache(
key=cache_key
@ -252,9 +254,7 @@ class _PROXY_MaxBudgetPerSessionHandler(CustomLogger):
return await self._in_memory_increment_spend(cache_key, amount)
async def _in_memory_increment_spend(
self, cache_key: str, amount: float
) -> float:
async def _in_memory_increment_spend(self, cache_key: str, amount: float) -> float:
current = await self.internal_usage_cache.async_get_cache(
key=cache_key,
litellm_parent_otel_span=None,

View file

@ -932,6 +932,26 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
)
)
# Session-level rate limits (per agent + session_id)
if _agent is not None and (
_agent.session_tpm_limit is not None
or _agent.session_rpm_limit is not None
):
_metadata = data.get("metadata") or data.get("litellm_metadata") or {}
_session_id = _metadata.get("session_id")
if _session_id is not None:
descriptors.append(
RateLimitDescriptor(
key="agent_session",
value=f"{_agent_id}:{_session_id}",
rate_limit={
"requests_per_unit": _agent.session_rpm_limit,
"tokens_per_unit": _agent.session_tpm_limit,
"window_size": self.window_size,
},
)
)
# Model rate limits
requested_model = data.get("model", None)
self._add_model_per_key_rate_limit_descriptor(
@ -1405,7 +1425,9 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
return "total" # default to total
return specified_rate_limit_type
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
async def async_log_success_event(
self, kwargs, response_obj, start_time, end_time
): # noqa: PLR0915
"""
Update TPM usage on successful API calls by incrementing counters using pipeline
"""
@ -1526,7 +1548,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
)
)
# Agent TPM
# Agent TPM + Agent session TPM
user_api_key_agent_id = standard_logging_metadata.get(
"user_api_key_agent_id"
)
@ -1539,6 +1561,20 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
total_tokens=total_tokens,
)
)
# Per-session token tracking (keyed by agent_id:session_id)
_lp = kwargs.get("litellm_params") or {}
_session_id = (
_lp.get("metadata") or _lp.get("litellm_metadata") or {}
).get("session_id")
if _session_id:
pipeline_operations.extend(
self._create_pipeline_operations(
key="agent_session",
value=f"{user_api_key_agent_id}:{_session_id}",
rate_limit_type="tokens",
total_tokens=total_tokens,
)
)
# Model-specific TPM
if model_group and user_api_key:

View file

@ -205,6 +205,15 @@ const AgentInfoView: React.FC<AgentInfoViewProps> = ({
<Descriptions.Item label="RPM Limit">{agent.rpm_limit ?? "Unlimited"}</Descriptions.Item>
<Descriptions.Item label="Session TPM Limit">{agent.session_tpm_limit ?? "Unlimited"}</Descriptions.Item>
<Descriptions.Item label="Session RPM Limit">{agent.session_rpm_limit ?? "Unlimited"}</Descriptions.Item>
{agent.spend !== undefined && agent.spend !== null && (
<Descriptions.Item label="Total Spend">${agent.spend.toFixed(6)}</Descriptions.Item>
)}
{agent.litellm_params?.max_iterations !== undefined && (
<Descriptions.Item label="Max Iterations">{agent.litellm_params.max_iterations}</Descriptions.Item>
)}
{agent.litellm_params?.max_budget_per_session !== undefined && (
<Descriptions.Item label="Max Budget Per Session">${agent.litellm_params.max_budget_per_session}</Descriptions.Item>
)}
<Descriptions.Item label="Created At">{formatDate(agent.created_at)}</Descriptions.Item>
<Descriptions.Item label="Updated At">{formatDate(agent.updated_at)}</Descriptions.Item>
</Descriptions>