Merge branch 'BerriAI:main' into fix/ollama-gpt-oss-thinking-field

This commit is contained in:
Cole McIntosh 2025-08-08 15:06:30 -06:00 • committed by GitHub
commit 66cc88ffb4
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
84 changed files with 1877 additions and 617 deletions

View file

@ -163,6 +163,14 @@ os.environ["OPENAI_BASE_URL"] = "https://your_host/v1" # OPTIONAL
| Model Name | Function Call |
|-----------------------|-----------------------------------------------------------------|
| gpt-5 | `response = completion(model="gpt-5", messages=messages)` |
| gpt-5-mini | `response = completion(model="gpt-5-mini", messages=messages)` |
| gpt-5-nano | `response = completion(model="gpt-5-nano", messages=messages)` |
| gpt-5-chat | `response = completion(model="gpt-5-chat", messages=messages)` |
| gpt-5-chat-latest | `response = completion(model="gpt-5-chat-latest", messages=messages)` |
| gpt-5-2025-08-07 | `response = completion(model="gpt-5-2025-08-07", messages=messages)` |
| gpt-5-mini-2025-08-07 | `response = completion(model="gpt-5-mini-2025-08-07", messages=messages)` |
| gpt-5-nano-2025-08-07 | `response = completion(model="gpt-5-nano-2025-08-07", messages=messages)` |
| gpt-4.1 | `response = completion(model="gpt-4.1", messages=messages)` |
| gpt-4.1-mini | `response = completion(model="gpt-4.1-mini", messages=messages)` |
| gpt-4.1-nano | `response = completion(model="gpt-4.1-nano", messages=messages)` |

View file

@ -12,7 +12,7 @@ import TabItem from '@theme/TabItem';
| Provider | [Microsoft Presidio](https://github.com/microsoft/presidio/) |
| Supported Entity Types | All Presidio Entity Types |
| Supported Actions | `MASK`, `BLOCK` |
| Supported Modes | `pre_call`, `during_call`, `post_call`, `logging_only` |
| Supported Modes | `pre_call`, `during_call`, `post_call`, `logging_only`, `pre_mcp_call` |
| Language Support | Configurable via `presidio_language` parameter (supports multiple languages including English, Spanish, German, etc.) |
## Deployment options
@ -239,7 +239,7 @@ guardrails:
- guardrail_name: "presidio-mask-guard"
litellm_params:
guardrail: presidio
mode: "pre_call"
mode: "pre_mcp_call" # Use this mode for MCP requests
pii_entities_config:
CREDIT_CARD: "MASK" # Will mask credit card numbers
EMAIL_ADDRESS: "MASK" # Will mask email addresses
@ -247,7 +247,7 @@ guardrails:
- guardrail_name: "presidio-block-guard"
litellm_params:
guardrail: presidio
mode: "pre_call"
mode: "pre_call" # Use this mode for regular LLM requests
pii_entities_config:
CREDIT_CARD: "BLOCK" # Will block requests containing credit card numbers
```
@ -338,6 +338,52 @@ The exception includes the entity type that was blocked (`CREDIT_CARD` in this c
## Advanced
### Supported Modes
The Presidio guardrail supports the following modes:
- `pre_call`: Run **before** LLM call, on **input**
- `post_call`: Run **after** LLM call, on **input & output**
- `logging_only`: Run **after** LLM call, only apply PII Masking before logging to Langfuse, etc. Not on the actual llm api request / response
- `pre_mcp_call`: Run **before** MCP call, on **input**. Use this mode when you want to apply PII masking/blocking for MCP requests
### MCP Usage Example
Here's how to use Presidio guardrails with MCP:
```yaml title="MCP Configuration Example" showLineNumbers
guardrails:
- guardrail_name: "presidio-mcp-guard"
litellm_params:
guardrail: presidio
mode: "pre_mcp_call"
pii_entities_config:
CREDIT_CARD: "MASK" # Will mask credit card numbers
EMAIL_ADDRESS: "BLOCK" # Will block email addresses
PHONE_NUMBER: "MASK" # Will mask phone numbers
MEDICAL_LICENSE: "BLOCK" # Will block medical license numbers
default_on: true
```
Test the MCP guardrail with a request:
```shell title="Test MCP Guardrail" showLineNumbers
curl http://localhost:4000/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer sk-1234" \
-d '{
"model": "gpt-3.5-turbo",
"messages": [
{"role": "user", "content": "My credit card is 4111-1111-1111-1111 and my medical license is ABC123"}
],
"guardrails": ["presidio-mcp-guard"]
}'
```
The request will be processed as follows:
1. Credit card number will be masked (e.g., replaced with `<CREDIT_CARD>`)
2. If a medical license is detected, the request will be blocked with a `BlockedPiiEntityError`
### Set `language` per request
The Presidio API [supports passing the `language` param](https://microsoft.github.io/presidio/api-docs/api-docs.html#tag/Analyzer/paths/~1analyze/post). Here is how to set the `language` per request

View file

@ -357,7 +357,7 @@ disable_copilot_system_to_assistant: bool = (
)
public_model_groups: Optional[List[str]] = None
public_model_groups_links: Dict[str, str] = {}
#### REQUEST PRIORITIZATION #####
#### REQUEST PRIORITIZATION ######
priority_reservation: Optional[Dict[str, float]] = None
@ -1145,6 +1145,9 @@ openaiOSeriesConfig = OpenAIOSeriesConfig()
from .llms.openai.chat.gpt_transformation import (
OpenAIGPTConfig,
)
from .llms.openai.chat.gpt_5_transformation import (
OpenAIGPT5Config,
)
from .llms.openai.transcriptions.whisper_transformation import (
OpenAIWhisperAudioTranscriptionConfig,
)
@ -1158,6 +1161,7 @@ from .llms.openai.chat.gpt_audio_transformation import (
)
openAIGPTAudioConfig = OpenAIGPTAudioConfig()
openAIGPT5Config = OpenAIGPT5Config()
from .llms.nvidia_nim.chat.transformation import NvidiaNimConfig
from .llms.nvidia_nim.embed import NvidiaNimEmbeddingConfig

View file

@ -4106,7 +4106,16 @@ class StandardLoggingPayloadSetup:
from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler
# Only generate object key if cold storage is configured
configured_cold_storage_logger = ColdStorageHandler._get_configured_cold_storage_custom_logger()
try:
configured_cold_storage_logger = (
ColdStorageHandler._get_configured_cold_storage_custom_logger()
)
except Exception as e:
verbose_logger.debug(
f"Cold storage custom logger unavailable: {e}"
)
return None
if configured_cold_storage_logger is None:
return None

View file

@ -0,0 +1,64 @@
"""Support for OpenAI gpt-5 model family."""
from typing import Optional
import litellm
from .gpt_transformation import OpenAIGPTConfig
class OpenAIGPT5Config(OpenAIGPTConfig):
"""Configuration for gpt-5 models.
Handles OpenAI API quirks for the gpt-5 series like:
- Mapping ``max_tokens`` -> ``max_completion_tokens``.
- Dropping unsupported ``temperature`` values when requested.
"""
@classmethod
def is_model_gpt_5_model(cls, model: str) -> bool:
return "gpt-5" in model
def get_supported_openai_params(self, model: str) -> list:
base_gpt_series_params = super().get_supported_openai_params(model=model)
gpt_5_only_params = ["reasoning_effort"]
base_gpt_series_params.extend(gpt_5_only_params)
return base_gpt_series_params
def map_openai_params(
self,
non_default_params: dict,
optional_params: dict,
model: str,
drop_params: bool,
) -> dict:
################################################################
# max_tokens is not supported for gpt-5 models on OpenAI API
# Relevant issue: https://github.com/BerriAI/litellm/issues/13381
################################################################
if "max_tokens" in non_default_params:
optional_params["max_completion_tokens"] = non_default_params.pop(
"max_tokens"
)
if "temperature" in non_default_params:
temperature_value: Optional[float] = non_default_params.pop("temperature")
if temperature_value is not None:
if temperature_value == 1:
optional_params["temperature"] = temperature_value
elif litellm.drop_params or drop_params:
pass
else:
raise litellm.utils.UnsupportedParamsError(
message=(
"gpt-5 models don't support temperature={}. Only temperature=1 is supported. To drop unsupported params set `litellm.drop_params = True`"
).format(temperature_value),
status_code=400,
)
return super()._map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
)

View file

@ -47,6 +47,7 @@ from litellm.utils import (
from ...types.llms.openai import *
from ..base import BaseLLM
from .chat.gpt_5_transformation import OpenAIGPT5Config
from .chat.o_series_transformation import OpenAIOSeriesConfig
from .common_utils import (
BaseOpenAILLM,
@ -55,6 +56,7 @@ from .common_utils import (
)
openaiOSeriesConfig = OpenAIOSeriesConfig()
openAIGPT5Config = OpenAIGPT5Config()
class MistralEmbeddingConfig:
@ -183,6 +185,8 @@ class OpenAIConfig(BaseConfig):
"""
if openaiOSeriesConfig.is_model_o_series_model(model=model):
return openaiOSeriesConfig.get_supported_openai_params(model=model)
elif openAIGPT5Config.is_model_gpt_5_model(model=model):
return openAIGPT5Config.get_supported_openai_params(model=model)
elif litellm.openAIGPTAudioConfig.is_model_gpt_audio_model(model=model):
return litellm.openAIGPTAudioConfig.get_supported_openai_params(model=model)
else:
@ -217,6 +221,13 @@ class OpenAIConfig(BaseConfig):
model=model,
drop_params=drop_params,
)
elif openAIGPT5Config.is_model_gpt_5_model(model=model):
return openAIGPT5Config.map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
)
elif litellm.openAIGPTAudioConfig.is_model_gpt_audio_model(model=model):
return litellm.openAIGPTAudioConfig.map_openai_params(
non_default_params=non_default_params,

View file

@ -614,7 +614,7 @@
},
"gpt-5": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -646,7 +646,7 @@
},
"gpt-5-mini": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -678,7 +678,7 @@
},
"gpt-5-nano": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -709,14 +709,12 @@
"supports_reasoning": true
},
"gpt-5-chat": {
"max_tokens": 32768,
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2e-05,
"input_cost_per_token_batches": 2.5e-06,
"output_cost_per_token_batches": 1e-05,
"cache_read_input_token_cost": 1.25e-06,
"max_tokens": 128000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
"cache_read_input_token_cost": 1.25e-07,
"litellm_provider": "openai",
"mode": "chat",
"supported_endpoints": [
@ -739,11 +737,12 @@
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_native_streaming": true
"supports_native_streaming": true,
"supports_reasoning": true
},
"gpt-5-chat-latest": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -775,7 +774,7 @@
},
"gpt-5-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 2720000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -807,7 +806,7 @@
},
"gpt-5-mini-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -839,7 +838,7 @@
},
"gpt-5-nano-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -2266,7 +2265,7 @@
},
"azure/gpt-5": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -2298,7 +2297,7 @@
},
"azure/gpt-5-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -2330,7 +2329,7 @@
},
"azure/gpt-5-mini": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -2362,7 +2361,7 @@
},
"azure/gpt-5-mini-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -2394,7 +2393,7 @@
},
"azure/gpt-5-nano-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -2426,7 +2425,7 @@
},
"azure/gpt-5-nano": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -2457,14 +2456,12 @@
"supports_reasoning": true
},
"azure/gpt-5-chat": {
"max_tokens": 32768,
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2e-05,
"input_cost_per_token_batches": 2.5e-06,
"output_cost_per_token_batches": 1e-05,
"cache_read_input_token_cost": 1.25e-06,
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
"cache_read_input_token_cost": 1.25e-07,
"litellm_provider": "azure",
"mode": "chat",
"supported_endpoints": [
@ -2487,11 +2484,13 @@
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_native_streaming": true
"supports_native_streaming": true,
"supports_reasoning": true,
"source": "https://azure.microsoft.com/en-us/blog/gpt-5-in-azure-ai-foundry-the-future-of-ai-apps-and-agents-starts-here/"
},
"azure/gpt-5-chat-latest": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -18244,7 +18243,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-4-scout-17b-16e-instruct": {
@ -18257,7 +18255,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-3.3-70b-instruct": {
@ -18270,7 +18267,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-3.2-90b-vision-instruct": {
@ -18283,7 +18279,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-3.1-405b-instruct": {
@ -18296,7 +18291,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
@ -18310,7 +18304,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3": {
@ -18323,7 +18316,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3-mini": {
@ -18336,7 +18328,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3-fast": {
@ -18349,7 +18340,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3-mini-fast": {
@ -18362,7 +18352,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
}
}

View file

@ -64,7 +64,7 @@ if MCP_AVAILABLE:
global_mcp_tool_registry,
)
from litellm.proxy._experimental.mcp_server.utils import (
get_server_name_prefix_tool_mcp,
get_server_name_prefix_tool_mcp,
)
######################################################
@ -127,9 +127,7 @@ if MCP_AVAILABLE:
await _sse_session_manager_cm.__aenter__()
_SESSION_MANAGERS_INITIALIZED = True
verbose_logger.info(
"MCP Server started with StreamableHTTP and SSE session managers!"
)
verbose_logger.info("MCP Server started with StreamableHTTP and SSE session managers!")
async def shutdown_session_managers():
"""Shutdown the session managers."""
@ -170,13 +168,11 @@ if MCP_AVAILABLE:
"""
try:
# Get user authentication from context variable
user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers, mcp_protocol_version = get_auth_context()
verbose_logger.debug(
f"MCP list_tools - User API Key Auth from context: {user_api_key_auth}"
)
verbose_logger.debug(
f"MCP list_tools - MCP servers from context: {mcp_servers}"
user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers, mcp_protocol_version = (
get_auth_context()
)
verbose_logger.debug(f"MCP list_tools - User API Key Auth from context: {user_api_key_auth}")
verbose_logger.debug(f"MCP list_tools - MCP servers from context: {mcp_servers}")
verbose_logger.debug(
f"MCP list_tools - MCP server auth headers: {list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None}"
)
@ -223,9 +219,7 @@ if MCP_AVAILABLE:
# Validate arguments
user_api_key_auth, mcp_auth_header, _, mcp_server_auth_headers, mcp_protocol_version = get_auth_context()
verbose_logger.debug(
f"MCP mcp_server_tool_call - User API Key Auth from context: {user_api_key_auth}"
)
verbose_logger.debug(f"MCP mcp_server_tool_call - User API Key Auth from context: {user_api_key_auth}")
try:
# Create a body date for logging
body_data = {"name": name, "arguments": arguments}
@ -258,31 +252,19 @@ if MCP_AVAILABLE:
except BlockedPiiEntityError as e:
verbose_logger.error(f"BlockedPiiEntityError in MCP tool call: {str(e)}")
# Return error as text content for MCP protocol
return [TextContent(
text=f"Error: Blocked PII entity detected - {str(e)}",
type="text"
)]
return [TextContent(text=f"Error: Blocked PII entity detected - {str(e)}", type="text")]
except GuardrailRaisedException as e:
verbose_logger.error(f"GuardrailRaisedException in MCP tool call: {str(e)}")
# Return error as text content for MCP protocol
return [TextContent(
text=f"Error: Guardrail violation - {str(e)}",
type="text"
)]
return [TextContent(text=f"Error: Guardrail violation - {str(e)}", type="text")]
except HTTPException as e:
verbose_logger.error(f"HTTPException in MCP tool call: {str(e)}")
# Return error as text content for MCP protocol
return [TextContent(
text=f"Error: {str(e.detail)}",
type="text"
)]
return [TextContent(text=f"Error: {str(e.detail)}", type="text")]
except Exception as e:
verbose_logger.exception(f"MCP mcp_server_tool_call - error: {e}")
# Return error as text content for MCP protocol
return [TextContent(
text=f"Error: {str(e)}",
type="text"
)]
return [TextContent(text=f"Error: {str(e)}", type="text")]
return response
@ -317,29 +299,40 @@ if MCP_AVAILABLE:
return []
# Get allowed MCP servers based on user permissions
allowed_mcp_servers = await global_mcp_server_manager.get_allowed_mcp_servers(
user_api_key_auth
)
allowed_mcp_servers = await global_mcp_server_manager.get_allowed_mcp_servers(user_api_key_auth)
filtered_server_ids = set()
# Filter servers based on mcp_servers parameter if provided
if mcp_servers is not None:
# Convert to lowercase for case-insensitive comparison
mcp_servers_lower = [s.lower() for s in mcp_servers]
allowed_mcp_servers = [
server_id
for server_id in allowed_mcp_servers
if any(
server_alias.lower() in mcp_servers_lower
for server in [global_mcp_server_manager.get_mcp_server_by_id(server_id)]
if server is not None
for server_alias in [
server.alias,
server.server_name,
server_id,
]
if server_alias is not None
)
]
for server_or_group in mcp_servers:
server_name_matched = False
for server_id in allowed_mcp_servers:
server = global_mcp_server_manager.get_mcp_server_by_id(server_id)
if server:
match_list = [s.lower() for s in [server.alias, server.server_name, server_id] if s is not None]
if server_or_group.lower() in match_list:
filtered_server_ids.add(server_id)
server_name_matched = True
break
if not server_name_matched:
try:
access_group_server_ids = await MCPRequestHandler._get_mcp_servers_from_access_groups(
[server_or_group]
)
# Only include servers that the user has access to
for server_id in access_group_server_ids:
if server_id in allowed_mcp_servers:
filtered_server_ids.add(server_id)
except Exception as e:
verbose_logger.debug(f"Could not resolve '{server_or_group}' as access group: {e}")
if filtered_server_ids:
allowed_mcp_servers = list(filtered_server_ids)
# Get tools from each allowed server
all_tools = []
@ -354,7 +347,7 @@ if MCP_AVAILABLE:
server_auth_header = mcp_server_auth_headers.get(server.alias)
elif mcp_server_auth_headers and server.server_name is not None:
server_auth_header = mcp_server_auth_headers.get(server.server_name)
# Fall back to deprecated mcp_auth_header if no server-specific header found
if server_auth_header is None:
server_auth_header = mcp_auth_header
@ -368,9 +361,7 @@ if MCP_AVAILABLE:
all_tools.extend(tools)
verbose_logger.debug(f"Successfully fetched {len(tools)} tools from server {server.name}")
except Exception as e:
verbose_logger.exception(
f"Error getting tools from server {server.name}: {str(e)}"
)
verbose_logger.exception(f"Error getting tools from server {server.name}: {str(e)}")
# Continue with other servers instead of failing completely
verbose_logger.info(f"Successfully fetched {len(all_tools)} tools total from all MCP servers")
@ -416,15 +407,11 @@ if MCP_AVAILABLE:
local_tools = []
try:
local_tools_raw = global_mcp_tool_registry.list_tools()
# Convert local tools to MCPTool format
for tool in local_tools_raw:
# Convert from litellm.types.mcp_server.tool_registry.MCPTool to mcp.types.Tool
mcp_tool = MCPTool(
name=tool.name,
description=tool.description,
inputSchema=tool.input_schema
)
mcp_tool = MCPTool(name=tool.name, description=tool.description, inputSchema=tool.input_schema)
local_tools.append(mcp_tool)
except Exception as e:
verbose_logger.exception(f"Error getting tools from local registry: {str(e)}")
@ -437,54 +424,42 @@ if MCP_AVAILABLE:
@client
async def call_mcp_tool(
name: str,
arguments: Optional[Dict[str, Any]] = None,
user_api_key_auth: Optional[UserAPIKeyAuth] = None,
mcp_auth_header: Optional[str] = None,
mcp_server_auth_headers: Optional[Dict[str, str]] = None,
mcp_protocol_version: Optional[str] = None,
**kwargs: Any
name: str,
arguments: Optional[Dict[str, Any]] = None,
user_api_key_auth: Optional[UserAPIKeyAuth] = None,
mcp_auth_header: Optional[str] = None,
mcp_server_auth_headers: Optional[Dict[str, str]] = None,
mcp_protocol_version: Optional[str] = None,
**kwargs: Any,
) -> List[Union[TextContent, ImageContent, EmbeddedResource]]:
"""
Call a specific tool with the provided arguments (handles prefixed tool names)
"""
start_time = datetime.now()
if arguments is None:
raise HTTPException(
status_code=400, detail="Request arguments are required"
)
raise HTTPException(status_code=400, detail="Request arguments are required")
# Remove prefix from tool name for logging and processing
original_tool_name, server_name_from_prefix = get_server_name_prefix_tool_mcp(
name
)
original_tool_name, server_name_from_prefix = get_server_name_prefix_tool_mcp(name)
standard_logging_mcp_tool_call: StandardLoggingMCPToolCall = (
_get_standard_logging_mcp_tool_call(
name=original_tool_name, # Use original name for logging
arguments=arguments,
server_name=server_name_from_prefix,
)
)
litellm_logging_obj: Optional[LiteLLMLoggingObj] = kwargs.get(
"litellm_logging_obj", None
standard_logging_mcp_tool_call: StandardLoggingMCPToolCall = _get_standard_logging_mcp_tool_call(
name=original_tool_name, # Use original name for logging
arguments=arguments,
server_name=server_name_from_prefix,
)
litellm_logging_obj: Optional[LiteLLMLoggingObj] = kwargs.get("litellm_logging_obj", None)
if litellm_logging_obj:
litellm_logging_obj.model_call_details["mcp_tool_call_metadata"] = (
standard_logging_mcp_tool_call
)
litellm_logging_obj.model_call_details["mcp_tool_call_metadata"] = standard_logging_mcp_tool_call
litellm_logging_obj.model = f"MCP: {name}"
# Try managed server tool first (pass the full prefixed name)
# Primary and recommended way to use MCP servers
#########################################################
mcp_server: Optional[MCPServer] = (
global_mcp_server_manager._get_mcp_server_from_tool_name(name)
)
mcp_server: Optional[MCPServer] = global_mcp_server_manager._get_mcp_server_from_tool_name(name)
if mcp_server:
standard_logging_mcp_tool_call["mcp_server_cost_info"] = (
mcp_server.mcp_info or {}
).get("mcp_server_cost_info")
response = await _handle_managed_mcp_tool(
standard_logging_mcp_tool_call["mcp_server_cost_info"] = (mcp_server.mcp_info or {}).get(
"mcp_server_cost_info"
)
response = await _handle_managed_mcp_tool(
name=name, # Pass the full name (potentially prefixed)
arguments=arguments,
user_api_key_auth=user_api_key_auth,
@ -500,7 +475,7 @@ if MCP_AVAILABLE:
#########################################################
else:
response = await _handle_local_mcp_tool(original_tool_name, arguments)
#########################################################
# Post MCP Tool Call Hook
# Allow modifying the MCP tool call response before it is returned to the user
@ -549,7 +524,7 @@ if MCP_AVAILABLE:
"""Handle tool execution for managed server tools"""
# Import here to avoid circular import
from litellm.proxy.proxy_server import proxy_logging_obj
call_tool_result = await global_mcp_server_manager.call_tool(
name=name,
arguments=arguments,
@ -584,6 +559,7 @@ if MCP_AVAILABLE:
Returns: (user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers)
"""
import re
mcp_servers_from_path = None
mcp_path_match = re.match(r"^/mcp/([^/]+)(/.*)?$", path)
if mcp_path_match:
@ -592,25 +568,39 @@ if MCP_AVAILABLE:
mcp_servers_from_path = [s.strip() for s in mcp_servers_str.split(",") if s.strip()]
if mcp_servers_from_path is not None:
user_api_key_auth, mcp_auth_header, _, mcp_server_auth_headers, mcp_protocol_version = (
await MCPRequestHandler.process_mcp_request(scope)
)
(
user_api_key_auth,
mcp_auth_header,
_,
mcp_server_auth_headers,
mcp_protocol_version,
) = await MCPRequestHandler.process_mcp_request(scope)
mcp_servers = mcp_servers_from_path
else:
user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers, mcp_protocol_version = (
await MCPRequestHandler.process_mcp_request(scope)
)
(
user_api_key_auth,
mcp_auth_header,
mcp_servers,
mcp_server_auth_headers,
mcp_protocol_version,
) = await MCPRequestHandler.process_mcp_request(scope)
return user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers, mcp_protocol_version
async def handle_streamable_http_mcp(
scope: Scope, receive: Receive, send: Send
) -> None:
async def handle_streamable_http_mcp(scope: Scope, receive: Receive, send: Send) -> None:
"""Handle MCP requests through StreamableHTTP."""
try:
path = scope.get("path", "")
user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers, mcp_protocol_version = await extract_mcp_auth_context(scope, path)
(
user_api_key_auth,
mcp_auth_header,
mcp_servers,
mcp_server_auth_headers,
mcp_protocol_version,
) = await extract_mcp_auth_context(scope, path)
verbose_logger.debug(f"MCP request mcp_servers (header/path): {mcp_servers}")
verbose_logger.debug(f"MCP server auth headers: {list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None}")
verbose_logger.debug(
f"MCP server auth headers: {list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None}"
)
verbose_logger.debug(f"MCP protocol version: {mcp_protocol_version}")
# Set the auth context variable for easy access in MCP functions
set_auth_context(
@ -635,10 +625,10 @@ if MCP_AVAILABLE:
# Send a proper HTTP error response instead of letting the exception bubble up
from starlette.responses import JSONResponse
from starlette.status import HTTP_500_INTERNAL_SERVER_ERROR
error_response = JSONResponse(
status_code=HTTP_500_INTERNAL_SERVER_ERROR,
content={"error": "MCP request failed", "details": str(e)}
content={"error": "MCP request failed", "details": str(e)},
)
await error_response(scope, receive, send)
except Exception as response_error:
@ -650,9 +640,17 @@ if MCP_AVAILABLE:
"""Handle MCP requests through SSE."""
try:
path = scope.get("path", "")
user_api_key_auth, mcp_auth_header, mcp_servers, mcp_server_auth_headers, mcp_protocol_version = await extract_mcp_auth_context(scope, path)
(
user_api_key_auth,
mcp_auth_header,
mcp_servers,
mcp_server_auth_headers,
mcp_protocol_version,
) = await extract_mcp_auth_context(scope, path)
verbose_logger.debug(f"MCP request mcp_servers (header/path): {mcp_servers}")
verbose_logger.debug(f"MCP server auth headers: {list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None}")
verbose_logger.debug(
f"MCP server auth headers: {list(mcp_server_auth_headers.keys()) if mcp_server_auth_headers else None}"
)
verbose_logger.debug(f"MCP protocol version: {mcp_protocol_version}")
set_auth_context(
user_api_key_auth=user_api_key_auth,
@ -674,10 +672,10 @@ if MCP_AVAILABLE:
# Send a proper HTTP error response instead of letting the exception bubble up
from starlette.responses import JSONResponse
from starlette.status import HTTP_500_INTERNAL_SERVER_ERROR
error_response = JSONResponse(
status_code=HTTP_500_INTERNAL_SERVER_ERROR,
content={"error": "MCP request failed", "details": str(e)}
content={"error": "MCP request failed", "details": str(e)},
)
await error_response(scope, receive, send)
except Exception as response_error:
@ -737,14 +735,14 @@ if MCP_AVAILABLE:
)
auth_context_var.set(auth_user)
def get_auth_context() -> (
Tuple[Optional[UserAPIKeyAuth], Optional[str], Optional[List[str]], Optional[Dict[str, str]], Optional[str]]
):
def get_auth_context() -> Tuple[
Optional[UserAPIKeyAuth], Optional[str], Optional[List[str]], Optional[Dict[str, str]], Optional[str]
]:
"""
Get the UserAPIKeyAuth from the auth context variable.
Returns:
Tuple[Optional[UserAPIKeyAuth], Optional[str], Optional[List[str]], Optional[Dict[str, str]]]:
Tuple[Optional[UserAPIKeyAuth], Optional[str], Optional[List[str]], Optional[Dict[str, str]]]:
UserAPIKeyAuth object, MCP auth header (deprecated), MCP servers (can include access groups), and server-specific auth headers
"""
auth_user = auth_context_var.get()

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[19813,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-9e1803a09e9ae8da.js","498","static/chunks/498-ee02f9b58491d7a9.js","154","static/chunks/154-78c3416dcb61977f.js","162","static/chunks/162-8529572226f208c5.js","172","static/chunks/172-08ae62d50ce1f0e7.js","931","static/chunks/app/page-0a9a9f137522a76c.js"],"default",1]
3:I[6691,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-3523e0e07cf314f6.js","683","static/chunks/683-07087d813e7eeb43.js","154","static/chunks/154-66d79df6143c694f.js","162","static/chunks/162-9e6f5133e328d61f.js","172","static/chunks/172-1c7afccd96ceca39.js","931","static/chunks/app/page-1d51309983956823.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[52829,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","154","static/chunks/154-78c3416dcb61977f.js","162","static/chunks/162-8529572226f208c5.js","418","static/chunks/app/model_hub/page-b26e0d313b582dbf.js"],"default",1]
3:I[52829,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","154","static/chunks/154-66d79df6143c694f.js","162","static/chunks/162-9e6f5133e328d61f.js","418","static/chunks/app/model_hub/page-b26e0d313b582dbf.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[22775,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-78c3416dcb61977f.js","162","static/chunks/162-8529572226f208c5.js","172","static/chunks/172-08ae62d50ce1f0e7.js","25","static/chunks/app/model_hub_table/page-d080c5775ebaf3a1.js"],"default",1]
3:I[22775,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-3523e0e07cf314f6.js","154","static/chunks/154-66d79df6143c694f.js","162","static/chunks/162-9e6f5133e328d61f.js","172","static/chunks/172-1c7afccd96ceca39.js","25","static/chunks/app/model_hub_table/page-d080c5775ebaf3a1.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","416","static/chunks/416-ad6bd55a20a586bd.js","154","static/chunks/154-78c3416dcb61977f.js","461","static/chunks/app/onboarding/page-883c32e6b072b842.js"],"default",1]
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","416","static/chunks/416-ad6bd55a20a586bd.js","154","static/chunks/154-66d79df6143c694f.js","461","static/chunks/app/onboarding/page-7e4cd2bb92dbf9ce.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

View file

@ -2164,6 +2164,7 @@ class AllCallbacks(LiteLLMPydanticObjectBase):
litellm_callback_params=[
"LANGFUSE_PUBLIC_KEY",
"LANGFUSE_SECRET_KEY",
"LANGFUSE_HOST",
],
)

View file

@ -267,6 +267,106 @@ async def get_api_key_metadata(
}
def _build_where_conditions(
*,
entity_id_field: str,
entity_id: Optional[Union[str, List[str]]],
start_date: str,
end_date: str,
model: Optional[str],
api_key: Optional[str],
exclude_entity_ids: Optional[List[str]] = None,
) -> Dict[str, Any]:
"""Build prisma where clause for daily activity queries."""
where_conditions: Dict[str, Any] = {
"date": {
"gte": start_date,
"lte": end_date,
}
}
if model:
where_conditions["model"] = model
if api_key:
where_conditions["api_key"] = api_key
if entity_id is not None:
if isinstance(entity_id, list):
where_conditions[entity_id_field] = {"in": entity_id}
else:
where_conditions[entity_id_field] = {"equals": entity_id}
if exclude_entity_ids:
current = where_conditions.get(entity_id_field, {})
if isinstance(current, str):
current = {"equals": current}
current["not"] = {"in": exclude_entity_ids}
where_conditions[entity_id_field] = current
return where_conditions
async def _aggregate_spend_records(
*,
prisma_client: PrismaClient,
records: List[Any],
entity_id_field: Optional[str],
entity_metadata_field: Optional[Dict[str, dict]],
) -> Dict[str, Any]:
"""Aggregate rows into DailySpendData list and total metrics."""
api_keys: Set[str] = set()
for record in records:
if record.api_key:
api_keys.add(record.api_key)
api_key_metadata: Dict[str, Dict[str, Any]] = {}
model_metadata: Dict[str, Dict[str, Any]] = {}
provider_metadata: Dict[str, Dict[str, Any]] = {}
if api_keys:
api_key_metadata = await get_api_key_metadata(prisma_client, api_keys)
results: List[DailySpendData] = []
total_metrics = SpendMetrics()
grouped_data: Dict[str, Dict[str, Any]] = {}
for record in records:
date_str = record.date
if date_str not in grouped_data:
grouped_data[date_str] = {
"metrics": SpendMetrics(),
"breakdown": BreakdownMetrics(),
}
grouped_data[date_str]["metrics"] = update_metrics(
grouped_data[date_str]["metrics"], record
)
grouped_data[date_str]["breakdown"] = update_breakdown_metrics(
grouped_data[date_str]["breakdown"],
record,
model_metadata,
provider_metadata,
api_key_metadata,
entity_id_field=entity_id_field,
entity_metadata_field=entity_metadata_field,
)
total_metrics = update_metrics(total_metrics, record)
for date_str, data in grouped_data.items():
results.append(
DailySpendData(
date=datetime.strptime(date_str, "%Y-%m-%d").date(),
metrics=data["metrics"],
breakdown=data["breakdown"],
)
)
results.sort(key=lambda x: x.date, reverse=True)
return {"results": results, "totals": total_metrics}
async def get_daily_activity(
prisma_client: Optional[PrismaClient],
table_name: str,
@ -296,27 +396,15 @@ async def get_daily_activity(
)
try:
# Build filter conditions
where_conditions: Dict[str, Any] = {
"date": {
"gte": start_date,
"lte": end_date,
}
}
if model:
where_conditions["model"] = model
if api_key:
where_conditions["api_key"] = api_key
if entity_id is not None:
if isinstance(entity_id, list):
where_conditions[entity_id_field] = {"in": entity_id}
else:
where_conditions[entity_id_field] = entity_id
if exclude_entity_ids:
where_conditions.setdefault(entity_id_field, {})["not"] = {
"in": exclude_entity_ids
}
where_conditions = _build_where_conditions(
entity_id_field=entity_id_field,
entity_id=entity_id,
start_date=start_date,
end_date=end_date,
model=model,
api_key=api_key,
exclude_entity_ids=exclude_entity_ids,
)
# Get total count for pagination
total_count = await getattr(prisma_client.db, table_name).count(
@ -333,87 +421,25 @@ async def get_daily_activity(
take=page_size,
)
# Get all unique API keys from the spend data
api_keys = set()
for record in daily_spend_data:
if record.api_key:
api_keys.add(record.api_key)
# Fetch key aliases in bulk
api_key_metadata: Dict[str, Dict[str, Any]] = {}
model_metadata: Dict[str, Dict[str, Any]] = {}
provider_metadata: Dict[str, Dict[str, Any]] = {}
if api_keys:
api_key_metadata = await get_api_key_metadata(prisma_client, api_keys)
# Process results
results = []
total_metrics = SpendMetrics()
grouped_data: Dict[str, Dict[str, Any]] = {}
for record in daily_spend_data:
date_str = record.date
if date_str not in grouped_data:
grouped_data[date_str] = {
"metrics": SpendMetrics(),
"breakdown": BreakdownMetrics(),
}
# Update metrics
grouped_data[date_str]["metrics"] = update_metrics(
grouped_data[date_str]["metrics"], record
)
# Update breakdowns
grouped_data[date_str]["breakdown"] = update_breakdown_metrics(
grouped_data[date_str]["breakdown"],
record,
model_metadata,
provider_metadata,
api_key_metadata,
entity_id_field=entity_id_field,
entity_metadata_field=entity_metadata_field,
)
# Update total metrics
total_metrics.spend += record.spend
total_metrics.prompt_tokens += record.prompt_tokens
total_metrics.completion_tokens += record.completion_tokens
total_metrics.total_tokens += (
record.prompt_tokens + record.completion_tokens
)
total_metrics.cache_read_input_tokens += record.cache_read_input_tokens
total_metrics.cache_creation_input_tokens += (
record.cache_creation_input_tokens
)
total_metrics.api_requests += record.api_requests
total_metrics.successful_requests += record.successful_requests
total_metrics.failed_requests += record.failed_requests
# Convert grouped data to response format
for date_str, data in grouped_data.items():
results.append(
DailySpendData(
date=datetime.strptime(date_str, "%Y-%m-%d").date(),
metrics=data["metrics"],
breakdown=data["breakdown"],
)
)
# Sort results by date
results.sort(key=lambda x: x.date, reverse=True)
aggregated = await _aggregate_spend_records(
prisma_client=prisma_client,
records=daily_spend_data,
entity_id_field=entity_id_field,
entity_metadata_field=entity_metadata_field,
)
return SpendAnalyticsPaginatedResponse(
results=results,
results=aggregated["results"],
metadata=DailySpendMetadata(
total_spend=total_metrics.spend,
total_prompt_tokens=total_metrics.prompt_tokens,
total_completion_tokens=total_metrics.completion_tokens,
total_tokens=total_metrics.total_tokens,
total_api_requests=total_metrics.api_requests,
total_successful_requests=total_metrics.successful_requests,
total_failed_requests=total_metrics.failed_requests,
total_cache_read_input_tokens=total_metrics.cache_read_input_tokens,
total_cache_creation_input_tokens=total_metrics.cache_creation_input_tokens,
total_spend=aggregated["totals"].spend,
total_prompt_tokens=aggregated["totals"].prompt_tokens,
total_completion_tokens=aggregated["totals"].completion_tokens,
total_tokens=aggregated["totals"].total_tokens,
total_api_requests=aggregated["totals"].api_requests,
total_successful_requests=aggregated["totals"].successful_requests,
total_failed_requests=aggregated["totals"].failed_requests,
total_cache_read_input_tokens=aggregated["totals"].cache_read_input_tokens,
total_cache_creation_input_tokens=aggregated["totals"].cache_creation_input_tokens,
page=page,
total_pages=-(-total_count // page_size), # Ceiling division
has_more=(page * page_size) < total_count,
@ -426,3 +452,85 @@ async def get_daily_activity(
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
detail={"error": f"Failed to fetch analytics: {str(e)}"},
)
async def get_daily_activity_aggregated(
prisma_client: Optional[PrismaClient],
table_name: str,
entity_id_field: str,
entity_id: Optional[Union[str, List[str]]],
entity_metadata_field: Optional[Dict[str, dict]],
start_date: Optional[str],
end_date: Optional[str],
model: Optional[str],
api_key: Optional[str],
exclude_entity_ids: Optional[List[str]] = None,
) -> SpendAnalyticsPaginatedResponse:
"""Aggregated variant that returns the full result set (no pagination).
Matches the response model of the paginated endpoint so the UI does not need to transform.
"""
if prisma_client is None:
raise HTTPException(
status_code=500,
detail={"error": CommonProxyErrors.db_not_connected_error.value},
)
if start_date is None or end_date is None:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={"error": "Please provide start_date and end_date"},
)
try:
where_conditions = _build_where_conditions(
entity_id_field=entity_id_field,
entity_id=entity_id,
start_date=start_date,
end_date=end_date,
model=model,
api_key=api_key,
exclude_entity_ids=exclude_entity_ids,
)
# Fetch all matching results (no pagination)
daily_spend_data = await getattr(prisma_client.db, table_name).find_many(
where=where_conditions,
order=[
{"date": "desc"},
],
)
aggregated = await _aggregate_spend_records(
prisma_client=prisma_client,
records=daily_spend_data,
entity_id_field=entity_id_field,
entity_metadata_field=entity_metadata_field,
)
return SpendAnalyticsPaginatedResponse(
results=aggregated["results"],
metadata=DailySpendMetadata(
total_spend=aggregated["totals"].spend,
total_prompt_tokens=aggregated["totals"].prompt_tokens,
total_completion_tokens=aggregated["totals"].completion_tokens,
total_tokens=aggregated["totals"].total_tokens,
total_api_requests=aggregated["totals"].api_requests,
total_successful_requests=aggregated["totals"].successful_requests,
total_failed_requests=aggregated["totals"].failed_requests,
total_cache_read_input_tokens=aggregated["totals"].cache_read_input_tokens,
total_cache_creation_input_tokens=aggregated["totals"].cache_creation_input_tokens,
page=1,
total_pages=1,
has_more=False,
),
)
except Exception as e:
verbose_proxy_logger.exception(
f"Error fetching aggregated daily activity: {str(e)}"
)
raise HTTPException(
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
detail={"error": f"Failed to fetch analytics: {str(e)}"},
)

View file

@ -26,7 +26,10 @@ from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import *
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.hooks.user_management_event_hooks import UserManagementEventHooks
from litellm.proxy.management_endpoints.common_daily_activity import get_daily_activity
from litellm.proxy.management_endpoints.common_daily_activity import (
get_daily_activity,
get_daily_activity_aggregated,
)
from litellm.proxy.management_endpoints.common_utils import _user_has_admin_view
from litellm.proxy.management_endpoints.key_management_endpoints import (
generate_key_helper_fn,
@ -35,13 +38,7 @@ from litellm.proxy.management_endpoints.key_management_endpoints import (
from litellm.proxy.management_helpers.utils import management_endpoint_wrapper
from litellm.proxy.utils import handle_exception_on_proxy
from litellm.types.proxy.management_endpoints.common_daily_activity import (
BreakdownMetrics,
KeyMetadata,
KeyMetricWithMetadata,
LiteLLM_DailyUserSpend,
MetricWithMetadata,
SpendAnalyticsPaginatedResponse,
SpendMetrics,
)
from litellm.types.proxy.management_endpoints.internal_user_endpoints import (
BulkUpdateUserRequest,
@ -1784,71 +1781,7 @@ async def ui_view_users(
raise HTTPException(status_code=500, detail=f"Error searching users: {str(e)}")
def update_metrics(
group_metrics: SpendMetrics, record: LiteLLM_DailyUserSpend
) -> SpendMetrics:
group_metrics.spend += record.spend
group_metrics.prompt_tokens += record.prompt_tokens
group_metrics.completion_tokens += record.completion_tokens
group_metrics.cache_read_input_tokens += record.cache_read_input_tokens
group_metrics.cache_creation_input_tokens += record.cache_creation_input_tokens
group_metrics.total_tokens += record.prompt_tokens + record.completion_tokens
group_metrics.api_requests += record.api_requests
group_metrics.successful_requests += record.successful_requests
group_metrics.failed_requests += record.failed_requests
return group_metrics
def update_breakdown_metrics(
breakdown: BreakdownMetrics,
record: LiteLLM_DailyUserSpend,
model_metadata: Dict[str, Dict[str, Any]],
provider_metadata: Dict[str, Dict[str, Any]],
api_key_metadata: Dict[str, Dict[str, Any]],
) -> BreakdownMetrics:
"""Updates breakdown metrics for a single record using the existing update_metrics function"""
# Update model breakdown
if record.model:
if record.model not in breakdown.models:
breakdown.models[record.model] = MetricWithMetadata(
metrics=SpendMetrics(),
metadata=model_metadata.get(
record.model, {}
), # Add any model-specific metadata here
)
breakdown.models[record.model].metrics = update_metrics(
breakdown.models[record.model].metrics, record
)
# Update provider breakdown
provider = record.custom_llm_provider or "unknown"
if provider not in breakdown.providers:
breakdown.providers[provider] = MetricWithMetadata(
metrics=SpendMetrics(),
metadata=provider_metadata.get(
provider, {}
), # Add any provider-specific metadata here
)
breakdown.providers[provider].metrics = update_metrics(
breakdown.providers[provider].metrics, record
)
# Update api key breakdown
if record.api_key not in breakdown.api_keys:
breakdown.api_keys[record.api_key] = KeyMetricWithMetadata(
metrics=SpendMetrics(),
metadata=KeyMetadata(
key_alias=api_key_metadata.get(record.api_key, {}).get(
"key_alias", None
)
), # Add any api_key-specific metadata here
)
breakdown.api_keys[record.api_key].metrics = update_metrics(
breakdown.api_keys[record.api_key].metrics, record
)
return breakdown
# Using shared metric helper implementations from common_daily_activity
@router.get(
@ -1857,6 +1790,7 @@ def update_breakdown_metrics(
dependencies=[Depends(user_api_key_auth)],
response_model=SpendAnalyticsPaginatedResponse,
)
@management_endpoint_wrapper
async def get_user_daily_activity(
start_date: Optional[str] = fastapi.Query(
default=None,
@ -1939,3 +1873,74 @@ async def get_user_daily_activity(
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
detail={"error": f"Failed to fetch analytics: {str(e)}"},
)
@router.get(
"/user/daily/activity/aggregated",
tags=["Budget & Spend Tracking", "Internal User management"],
dependencies=[Depends(user_api_key_auth)],
response_model=SpendAnalyticsPaginatedResponse,
)
@management_endpoint_wrapper
async def get_user_daily_activity_aggregated(
start_date: Optional[str] = fastapi.Query(
default=None,
description="Start date in YYYY-MM-DD format",
),
end_date: Optional[str] = fastapi.Query(
default=None,
description="End date in YYYY-MM-DD format",
),
model: Optional[str] = fastapi.Query(
default=None,
description="Filter by specific model",
),
api_key: Optional[str] = fastapi.Query(
default=None,
description="Filter by specific API key",
),
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
) -> SpendAnalyticsPaginatedResponse:
"""
Aggregated analytics for a user's daily activity without pagination.
Returns the same response shape as the paginated endpoint with page metadata set to single-page.
"""
from litellm.proxy.proxy_server import prisma_client
if prisma_client is None:
raise HTTPException(
status_code=500,
detail={"error": CommonProxyErrors.db_not_connected_error.value},
)
if start_date is None or end_date is None:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={"error": "Please provide start_date and end_date"},
)
try:
entity_id: Optional[str] = None
if not _user_has_admin_view(user_api_key_dict):
entity_id = user_api_key_dict.user_id
return await get_daily_activity_aggregated(
prisma_client=prisma_client,
table_name="litellm_dailyuserspend",
entity_id_field="user_id",
entity_id=entity_id,
entity_metadata_field=None,
start_date=start_date,
end_date=end_date,
model=model,
api_key=api_key,
)
except Exception as e:
verbose_proxy_logger.exception(
"/user/daily/activity/aggregated: Exception occured - {}".format(str(e))
)
raise HTTPException(
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
detail={"error": f"Failed to fetch analytics: {str(e)}"},
)

View file

@ -1434,6 +1434,20 @@ async def team_member_delete(
data={"teams": {"set": team_list}},
)
# Also clean up any existing team membership rows for this user and team
user_ids_to_delete = set()
if data.user_id is not None:
user_ids_to_delete.add(data.user_id)
if existing_user_rows is not None and isinstance(existing_user_rows, list):
for existing_user in existing_user_rows:
if getattr(existing_user, "user_id", None):
user_ids_to_delete.add(existing_user.user_id)
for _uid in user_ids_to_delete:
await prisma_client.db.litellm_teammembership.delete_many(
where={"team_id": data.team_id, "user_id": _uid}
)
return existing_team_row

View file

@ -1,15 +1,5 @@
model_list:
- model_name: bedrock/*
- model_name: gemini/*
litellm_params:
model: bedrock/*
model: gemini/*
litellm_settings:
callbacks: ["s3_v2"]
s3_callback_params:
s3_bucket_name: litellm-logs # AWS Bucket Name for S3
s3_region_name: us-west-2
general_settings:
cold_storage_custom_logger: s3_v2
store_prompts_in_cold_storage: true

View file

@ -8677,6 +8677,7 @@ async def get_config(): # noqa: PLR0915
elif _callback == "braintrust":
env_vars = [
"BRAINTRUST_API_KEY",
"BRAINTRUST_API_BASE",
]
elif _callback == "traceloop":
env_vars = ["TRACELOOP_API_KEY"]

View file

@ -64,10 +64,31 @@ class ColdStorageHandler:
@staticmethod
def _get_configured_cold_storage_custom_logger() -> Optional[_custom_logger_compatible_callbacks_literal]:
from litellm.proxy.proxy_server import general_settings
cold_storage_custom_logger: Optional[str] = general_settings.get("cold_storage_custom_logger")
if not cold_storage_custom_logger:
verbose_proxy_logger.debug("No cold storage custom logger found in general settings")
"""Return the configured cold storage custom logger.
During interpreter shutdown importing ``proxy_server`` can raise a
``RuntimeError`` (e.g. "can't register atexit after shutdown").
In these scenarios we gracefully return ``None`` instead of bubbling
the exception up the call stack.
"""
try:
from litellm.proxy.proxy_server import general_settings
except Exception as e:
verbose_proxy_logger.debug(
f"Unable to import proxy_server for cold storage logging: {e}"
)
return None
return cast(_custom_logger_compatible_callbacks_literal, cold_storage_custom_logger)
cold_storage_custom_logger: Optional[str] = general_settings.get(
"cold_storage_custom_logger"
)
if not cold_storage_custom_logger:
verbose_proxy_logger.debug(
"No cold storage custom logger found in general settings"
)
return None
return cast(
_custom_logger_compatible_callbacks_literal, cold_storage_custom_logger
)

View file

@ -84,6 +84,8 @@ class LiteLLMCompletionTransformationHandler:
litellm_custom_stream_wrapper=litellm_completion_response,
request_input=input,
responses_api_request=responses_api_request,
custom_llm_provider=custom_llm_provider,
litellm_metadata=kwargs.get("litellm_metadata", {}),
)
async def async_response_api_handler(
@ -129,4 +131,6 @@ class LiteLLMCompletionTransformationHandler:
litellm_custom_stream_wrapper=litellm_completion_response,
request_input=request_input,
responses_api_request=responses_api_request,
custom_llm_provider=litellm_completion_request.get("custom_llm_provider"),
litellm_metadata=kwargs.get("litellm_metadata", {}),
)

View file

@ -6,6 +6,7 @@ from litellm.responses.litellm_completion_transformation.transformation import (
LiteLLMCompletionResponsesConfig,
)
from litellm.responses.streaming_iterator import ResponsesAPIStreamingIterator
from litellm.responses.utils import ResponsesAPIRequestUtils
from litellm.types.llms.openai import (
OutputTextDeltaEvent,
ReasoningSummaryTextDeltaEvent,
@ -34,6 +35,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
litellm_custom_stream_wrapper: litellm.CustomStreamWrapper,
request_input: Union[str, ResponseInputParam],
responses_api_request: ResponsesAPIOptionalRequestParams,
custom_llm_provider: Optional[str] = None,
litellm_metadata: Optional[dict] = None,
):
self.litellm_custom_stream_wrapper: litellm.CustomStreamWrapper = (
litellm_custom_stream_wrapper
@ -42,6 +45,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
self.responses_api_request: ResponsesAPIOptionalRequestParams = (
responses_api_request
)
self.custom_llm_provider: Optional[str] = custom_llm_provider
self.litellm_metadata: Optional[dict] = litellm_metadata or {}
self.collected_chat_completion_chunks: List[ModelResponseStream] = []
self.finished: bool = False
@ -164,14 +169,23 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
Union[ModelResponse, TextCompletionResponse]
] = stream_chunk_builder(chunks=self.collected_chat_completion_chunks)
if litellm_model_response and isinstance(litellm_model_response, ModelResponse):
# Transform the response
responses_api_response = LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response(
request_input=self.request_input,
chat_completion_response=litellm_model_response,
responses_api_request=self.responses_api_request,
)
# Encode the response ID to match non-streaming behavior
encoded_response = ResponsesAPIRequestUtils._update_responses_api_response_id_with_model_id(
responses_api_response=responses_api_response,
custom_llm_provider=self.custom_llm_provider,
litellm_metadata=self.litellm_metadata,
)
return ResponseCompletedEvent(
type=ResponsesAPIStreamEvents.RESPONSE_COMPLETED,
response=LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response(
request_input=self.request_input,
chat_completion_response=litellm_model_response,
responses_api_request=self.responses_api_request,
),
response=encoded_response,
)
else:
return None

View file

@ -614,7 +614,7 @@
},
"gpt-5": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -646,7 +646,7 @@
},
"gpt-5-mini": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -678,7 +678,7 @@
},
"gpt-5-nano": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -709,14 +709,12 @@
"supports_reasoning": true
},
"gpt-5-chat": {
"max_tokens": 32768,
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2e-05,
"input_cost_per_token_batches": 2.5e-06,
"output_cost_per_token_batches": 1e-05,
"cache_read_input_token_cost": 1.25e-06,
"max_tokens": 128000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
"cache_read_input_token_cost": 1.25e-07,
"litellm_provider": "openai",
"mode": "chat",
"supported_endpoints": [
@ -739,11 +737,12 @@
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_native_streaming": true
"supports_native_streaming": true,
"supports_reasoning": true
},
"gpt-5-chat-latest": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -775,7 +774,7 @@
},
"gpt-5-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 2720000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -807,7 +806,7 @@
},
"gpt-5-mini-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -839,7 +838,7 @@
},
"gpt-5-nano-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -2266,7 +2265,7 @@
},
"azure/gpt-5": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -2298,7 +2297,7 @@
},
"azure/gpt-5-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -2330,7 +2329,7 @@
},
"azure/gpt-5-mini": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -2362,7 +2361,7 @@
},
"azure/gpt-5-mini-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 2.5e-07,
"output_cost_per_token": 2e-06,
@ -2394,7 +2393,7 @@
},
"azure/gpt-5-nano-2025-08-07": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -2426,7 +2425,7 @@
},
"azure/gpt-5-nano": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"input_cost_per_token": 5e-08,
"output_cost_per_token": 4e-07,
@ -2457,14 +2456,12 @@
"supports_reasoning": true
},
"azure/gpt-5-chat": {
"max_tokens": 32768,
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"input_cost_per_token": 5e-06,
"output_cost_per_token": 2e-05,
"input_cost_per_token_batches": 2.5e-06,
"output_cost_per_token_batches": 1e-05,
"cache_read_input_token_cost": 1.25e-06,
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
"cache_read_input_token_cost": 1.25e-07,
"litellm_provider": "azure",
"mode": "chat",
"supported_endpoints": [
@ -2487,11 +2484,13 @@
"supports_prompt_caching": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_native_streaming": true
"supports_native_streaming": true,
"supports_reasoning": true,
"source": "https://azure.microsoft.com/en-us/blog/gpt-5-in-azure-ai-foundry-the-future-of-ai-apps-and-agents-starts-here/"
},
"azure/gpt-5-chat-latest": {
"max_tokens": 128000,
"max_input_tokens": 400000,
"max_input_tokens": 128000,
"max_output_tokens": 128000,
"input_cost_per_token": 1.25e-06,
"output_cost_per_token": 1e-05,
@ -18244,7 +18243,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-4-scout-17b-16e-instruct": {
@ -18257,7 +18255,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-3.3-70b-instruct": {
@ -18270,7 +18267,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-3.2-90b-vision-instruct": {
@ -18283,7 +18279,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/meta.llama-3.1-405b-instruct": {
@ -18296,7 +18291,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
@ -18310,7 +18304,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3": {
@ -18323,7 +18316,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3-mini": {
@ -18336,7 +18328,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3-fast": {
@ -18349,7 +18340,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
},
"oci/xai.grok-3-mini-fast": {
@ -18362,7 +18352,6 @@
"mode": "chat",
"supports_function_calling": true,
"supports_response_schema": false,
"supports_tool_choice": false,
"source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing"
}
}

View file

@ -137,6 +137,14 @@ model_list:
model: openai/my-fake-model
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.appxxxx/
- model_name: gemini-1.5-flash
litellm_params:
model: gemini/gemini-1.5-flash
api_key: os.environ/GOOGLE_API_KEY
- model_name: gpt-4o
litellm_params:
model: gpt-4o
api_key: os.environ/OPENAI_API_KEY
litellm_settings:

View file

@ -1,6 +1,6 @@
[tool.poetry]
name = "litellm"
version = "1.75.2"
version = "1.75.3"
description = "Library to easily interface with LLM API providers"
authors = ["BerriAI"]
license = "MIT"
@ -154,7 +154,7 @@ requires = ["poetry-core", "wheel"]
build-backend = "poetry.core.masonry.api"
[tool.commitizen]
version = "1.75.2"
version = "1.75.3"
version_files = [
"pyproject.toml:^version"
]

View file

@ -0,0 +1,239 @@
"""
Unit tests for BaseResponsesAPIStreamingIterator
Tests core functionality including:
1. Processing chunks and handling ResponseCompletedEvent
2. Ensuring _update_responses_api_response_id_with_model_id is called for final chunk
3. Verifying ID update is NOT called for non-final chunks (delta events)
4. Edge case handling for invalid JSON, empty chunks, and [DONE] markers
These tests ensure the streaming iterator correctly processes response chunks
and applies model ID updates only to completed responses, as required for proper
response tracking and logging.
"""
import json
import os
import sys
from datetime import datetime
from typing import Any, Dict, Optional
from unittest.mock import Mock, patch
import pytest
sys.path.insert(0, os.path.abspath("../.."))
from litellm.constants import STREAM_SSE_DONE_STRING
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.base_llm.responses.transformation import BaseResponsesAPIConfig
from litellm.responses.streaming_iterator import BaseResponsesAPIStreamingIterator
from litellm.responses.utils import ResponsesAPIRequestUtils
from litellm.types.llms.openai import (
ResponseCompletedEvent,
ResponsesAPIResponse,
ResponsesAPIStreamEvents,
OutputTextDeltaEvent
)
class TestBaseResponsesAPIStreamingIterator:
"""Test cases for BaseResponsesAPIStreamingIterator"""
def test_process_chunk_with_response_completed_event(self):
"""
Test that _process_chunk correctly processes a ResponseCompletedEvent
and calls _update_responses_api_response_id_with_model_id for the final chunk.
"""
# Mock dependencies
mock_response = Mock()
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
mock_config = Mock(spec=BaseResponsesAPIConfig)
# Create a mock ResponsesAPIResponse for the completed event
mock_responses_api_response = Mock(spec=ResponsesAPIResponse)
mock_responses_api_response.id = "original_response_id"
# Create a mock ResponseCompletedEvent
mock_completed_event = Mock(spec=ResponseCompletedEvent)
mock_completed_event.type = ResponsesAPIStreamEvents.RESPONSE_COMPLETED
mock_completed_event.response = mock_responses_api_response
# Set up the mock transform method to return our completed event
mock_config.transform_streaming_response.return_value = mock_completed_event
# Mock the _update_responses_api_response_id_with_model_id method
updated_response = Mock(spec=ResponsesAPIResponse)
updated_response.id = "updated_response_id"
# Create the iterator instance
iterator = BaseResponsesAPIStreamingIterator(
response=mock_response,
model="gpt-4",
responses_api_provider_config=mock_config,
logging_obj=mock_logging_obj,
litellm_metadata={"model_info": {"id": "model_123"}},
custom_llm_provider="openai"
)
# Prepare test chunk data
test_chunk_data = {
"type": "response.completed",
"response": {
"id": "original_response_id",
"output": [{"type": "message", "content": [{"text": "Hello World"}]}]
}
}
with patch.object(
ResponsesAPIRequestUtils,
'_update_responses_api_response_id_with_model_id',
return_value=updated_response
) as mock_update_id:
# Process the chunk
result = iterator._process_chunk(json.dumps(test_chunk_data))
# Assertions
assert result is not None
assert result.type == ResponsesAPIStreamEvents.RESPONSE_COMPLETED
# Verify that _update_responses_api_response_id_with_model_id was called
mock_update_id.assert_called_once_with(
responses_api_response=mock_responses_api_response,
litellm_metadata={"model_info": {"id": "model_123"}},
custom_llm_provider="openai"
)
# Verify the completed response was stored
assert iterator.completed_response == result
# Verify the response was updated on the event
assert result.response == updated_response
def test_process_chunk_with_delta_event_no_id_update(self):
"""
Test that _process_chunk correctly processes a delta event
and does NOT call _update_responses_api_response_id_with_model_id.
"""
# Mock dependencies
mock_response = Mock()
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
mock_config = Mock(spec=BaseResponsesAPIConfig)
# Create a mock OutputTextDeltaEvent (not a completed event)
mock_delta_event = Mock(spec=OutputTextDeltaEvent)
mock_delta_event.type = ResponsesAPIStreamEvents.OUTPUT_TEXT_DELTA
mock_delta_event.delta = "Hello"
# Delta events don't have a response attribute
delattr(mock_delta_event, 'response') if hasattr(mock_delta_event, 'response') else None
# Set up the mock transform method to return our delta event
mock_config.transform_streaming_response.return_value = mock_delta_event
# Create the iterator instance
iterator = BaseResponsesAPIStreamingIterator(
response=mock_response,
model="gpt-4",
responses_api_provider_config=mock_config,
logging_obj=mock_logging_obj,
litellm_metadata={"model_info": {"id": "model_123"}},
custom_llm_provider="openai"
)
# Prepare test chunk data for a delta event
test_chunk_data = {
"type": "response.output_text.delta",
"delta": "Hello",
"item_id": "item_123",
"output_index": 0,
"content_index": 0
}
with patch.object(
ResponsesAPIRequestUtils,
'_update_responses_api_response_id_with_model_id'
) as mock_update_id:
# Process the chunk
result = iterator._process_chunk(json.dumps(test_chunk_data))
# Assertions
assert result is not None
assert result.type == ResponsesAPIStreamEvents.OUTPUT_TEXT_DELTA
# Verify that _update_responses_api_response_id_with_model_id was NOT called
mock_update_id.assert_not_called()
# Verify no completed response was stored (since this is not a completed event)
assert iterator.completed_response is None
def test_process_chunk_handles_invalid_json(self):
"""
Test that _process_chunk gracefully handles invalid JSON.
"""
# Mock dependencies
mock_response = Mock()
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
mock_config = Mock(spec=BaseResponsesAPIConfig)
# Create the iterator instance
iterator = BaseResponsesAPIStreamingIterator(
response=mock_response,
model="gpt-4",
responses_api_provider_config=mock_config,
logging_obj=mock_logging_obj
)
# Test with invalid JSON
result = iterator._process_chunk("invalid json {")
# Should return None for invalid JSON
assert result is None
assert iterator.completed_response is None
def test_process_chunk_handles_done_marker(self):
"""
Test that _process_chunk correctly handles the [DONE] marker.
"""
# Mock dependencies
mock_response = Mock()
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
mock_config = Mock(spec=BaseResponsesAPIConfig)
# Create the iterator instance
iterator = BaseResponsesAPIStreamingIterator(
response=mock_response,
model="gpt-4",
responses_api_provider_config=mock_config,
logging_obj=mock_logging_obj
)
# Test with [DONE] marker
result = iterator._process_chunk(STREAM_SSE_DONE_STRING)
# Should return None and set finished flag
assert result is None
assert iterator.finished is True
def test_process_chunk_handles_empty_chunk(self):
"""
Test that _process_chunk correctly handles empty or None chunks.
"""
# Mock dependencies
mock_response = Mock()
mock_logging_obj = Mock(spec=LiteLLMLoggingObj)
mock_config = Mock(spec=BaseResponsesAPIConfig)
# Create the iterator instance
iterator = BaseResponsesAPIStreamingIterator(
response=mock_response,
model="gpt-4",
responses_api_provider_config=mock_config,
logging_obj=mock_logging_obj
)
# Test with empty chunk
result = iterator._process_chunk("")
assert result is None
# Test with None chunk
result = iterator._process_chunk(None)
assert result is None

View file

@ -168,56 +168,6 @@ def test_stream_chunk_builder_litellm_tool_call_regular_message():
# test_stream_chunk_builder_litellm_tool_call_regular_message()
def test_stream_chunk_builder_litellm_usage_chunks():
"""
Checks if stream_chunk_builder is able to correctly rebuild with given metadata from streaming chunks
"""
from litellm.types.utils import Usage
messages = [
{"role": "user", "content": "Tell me the funniest joke you know."},
{
"role": "assistant",
"content": "Why did the chicken cross the road?\nYou will not guess this one I bet\n",
},
{"role": "user", "content": "I do not know, why?"},
{"role": "assistant", "content": "uhhhh\n\n\nhmmmm.....\nthinking....\n"},
{"role": "user", "content": "\nI am waiting...\n\n...\n"},
]
usage: litellm.Usage = Usage(
completion_tokens=27,
prompt_tokens=50,
total_tokens=82,
completion_tokens_details=None,
prompt_tokens_details=None,
)
gemini_pt = usage.prompt_tokens
# make a streaming gemini call
try:
response = completion(
model="gemini/gemini-2.5-flash-lite",
messages=messages,
stream=True,
complete_response=True,
stream_options={"include_usage": True},
)
except litellm.InternalServerError as e:
pytest.skip(f"Skipping test due to internal server error - {str(e)}")
usage: litellm.Usage = response.usage
stream_rebuilt_pt = usage.prompt_tokens
# assert prompt tokens are the same
assert (
gemini_pt == stream_rebuilt_pt
), f"Stream builder is not able to rebuild usage correctly. Got={stream_rebuilt_pt}, expected={gemini_pt}"
def test_stream_chunk_builder_litellm_mixed_calls():
response = stream_chunk_builder(stream_chunk_testdata.chunks)
assert (

View file

@ -698,6 +698,15 @@ async def test_get_tools_from_mcp_servers():
transport=MCPTransport.http,
spec_version=MCPSpecVersion.nov_2024
)
mock_server_3 = MCPServer(
server_id="server3_id",
name="server3",
server_name="server3",
url="http://test3.com",
transport=MCPTransport.http,
spec_version=MCPSpecVersion.nov_2024,
access_groups=["group-a"]
)
mock_tool_1 = MCPTool(name="tool1", description="test tool 1", inputSchema={})
mock_tool_2 = MCPTool(name="tool2", description="test tool 2", inputSchema={})
@ -709,6 +718,8 @@ async def test_get_tools_from_mcp_servers():
return mock_server_1
elif server_id == "server2_id":
return mock_server_2
elif server_id == "server3_id":
return mock_server_3
return None
# Create a mock manager
@ -744,6 +755,26 @@ async def test_get_tools_from_mcp_servers():
assert len(result) == 2, "Should return tools from all servers"
assert result[0].name == "tool1" and result[1].name == "tool2", "Should return tools from all servers"
#
# Test Case 3: With specific MCP servers and access groups
# Create a mock manager
mock_manager = AsyncMock()
mock_manager.get_allowed_mcp_servers = AsyncMock(return_value=["server1_id", "server2_id", "server3_id"])
mock_manager.get_mcp_server_by_id = mock_get_server_by_id
mock_manager._get_tools_from_server = AsyncMock(return_value=[mock_tool_1])
with patch('litellm.proxy._experimental.mcp_server.server.global_mcp_server_manager', mock_manager):
with patch('litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp.MCPRequestHandler._get_mcp_servers_from_access_groups', AsyncMock(return_value=["server3_id"])):
# Test with specific servers
result = await _get_tools_from_mcp_servers(
user_api_key_auth=mock_user_auth,
mcp_auth_header=mock_auth_header,
mcp_servers=["group-a"],
)
assert len(result) == 1, "Should only return tools from server3"
assert result[0].name == "tool1", "Should return tool from server1"
except AssertionError as e:
pytest.fail(f"Test failed: {str(e)}")
except Exception as e:

View file

@ -464,3 +464,30 @@ async def test_e2e_generate_cold_storage_object_key_not_configured():
# Verify the result is None when cold storage is not configured
assert result is None
@pytest.mark.asyncio
async def test_e2e_generate_cold_storage_object_key_runtime_error_handled():
"""Ensure runtime errors while loading cold storage logger are ignored."""
from datetime import datetime, timezone
from unittest.mock import patch
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
start_time = datetime(2025, 1, 15, 10, 30, 45, 123456, timezone.utc)
response_id = "chatcmpl-test-runtime"
team_alias = "team"
with patch(
"litellm.proxy.spend_tracking.cold_storage_handler.ColdStorageHandler._get_configured_cold_storage_custom_logger",
side_effect=RuntimeError("can't register atexit after shutdown"),
):
result = StandardLoggingPayloadSetup._generate_cold_storage_object_key(
start_time=start_time,
response_id=response_id,
team_alias=team_alias,
)
# When an exception occurs retrieving the cold storage logger, the
# function should return None instead of raising.
assert result is None

View file

@ -243,3 +243,85 @@ def test_cache_read_input_tokens_retained():
assert usage.cache_creation_input_tokens == 4
assert usage.cache_read_input_tokens == 11775
assert usage.prompt_tokens_details.cached_tokens == 11775
def test_stream_chunk_builder_litellm_usage_chunks():
"""
Validate ChunkProcessor.calculate_usage uses provided usage fields from streaming chunks
and reconstructs prompt and completion tokens without making any upstream API calls.
"""
# Prepare two mocked streaming chunks with usage split across them
chunk1 = ModelResponseStream(
id="chatcmpl-mocked-usage-1",
created=1745513206,
model="gemini/gemini-2.5-flash-lite",
object="chat.completion.chunk",
system_fingerprint=None,
choices=[
StreamingChoices(
finish_reason=None,
index=0,
delta=Delta(
provider_specific_fields=None,
content="",
role=None,
function_call=None,
tool_calls=None,
audio=None,
),
logprobs=None,
)
],
provider_specific_fields=None,
stream_options={"include_usage": True},
usage=Usage(
completion_tokens=0,
prompt_tokens=50,
total_tokens=50,
completion_tokens_details=None,
prompt_tokens_details=None,
),
)
chunk2 = ModelResponseStream(
id="chatcmpl-mocked-usage-1",
created=1745513207,
model="gemini/gemini-2.5-flash-lite",
object="chat.completion.chunk",
system_fingerprint=None,
choices=[
StreamingChoices(
finish_reason="stop",
index=0,
delta=Delta(
provider_specific_fields=None,
content=None,
role=None,
function_call=None,
tool_calls=None,
audio=None,
),
logprobs=None,
)
],
provider_specific_fields=None,
stream_options={"include_usage": True},
usage=Usage(
completion_tokens=27,
prompt_tokens=0,
total_tokens=27,
completion_tokens_details=None,
prompt_tokens_details=None,
),
)
chunks = [chunk1, chunk2]
processor = ChunkProcessor(chunks=chunks)
usage = processor.calculate_usage(
chunks=chunks, model="gemini/gemini-2.5-flash-lite", completion_output=""
)
assert usage.prompt_tokens == 50
assert usage.completion_tokens == 27
assert usage.total_tokens == 77

View file

@ -0,0 +1,43 @@
import pytest
import litellm
from litellm.llms.openai.openai import OpenAIConfig
@pytest.fixture()
def config() -> OpenAIConfig:
return OpenAIConfig()
def test_gpt5_supports_reasoning_effort(config: OpenAIConfig):
assert "reasoning_effort" in config.get_supported_openai_params(model="gpt-5")
assert "reasoning_effort" in config.get_supported_openai_params(model="gpt-5-mini")
def test_gpt5_maps_max_tokens(config: OpenAIConfig):
params = config.map_openai_params(
non_default_params={"max_tokens": 10},
optional_params={},
model="gpt-5",
drop_params=False,
)
assert params["max_completion_tokens"] == 10
assert "max_tokens" not in params
def test_gpt5_temperature_drop(config: OpenAIConfig):
params = config.map_openai_params(
non_default_params={"temperature": 0.2},
optional_params={},
model="gpt-5",
drop_params=True,
)
assert "temperature" not in params
def test_gpt5_temperature_error(config: OpenAIConfig):
with pytest.raises(litellm.utils.UnsupportedParamsError):
config.map_openai_params(
non_default_params={"temperature": 0.2},
optional_params={},
model="gpt-5",
drop_params=False,
)

View file

@ -762,7 +762,7 @@ async def test_validate_team_member_add_permissions_admin():
)
# Create admin user
admin_user = UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN.value)
admin_user = UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN)
# Create mock team
team = MagicMock(spec=LiteLLM_TeamTable)
@ -787,7 +787,7 @@ async def test_validate_team_member_add_permissions_non_admin():
# Create non-admin user
regular_user = UserAPIKeyAuth(
user_id="regular-user",
user_role=LitellmUserRoles.INTERNAL_USER.value,
user_role=LitellmUserRoles.INTERNAL_USER,
team_id="different-team",
)
@ -886,8 +886,8 @@ async def test_process_team_members_multiple_members():
# Create multiple members as dictionaries (they will be converted to Member objects)
members = [
{"user_email": "user1@example.com", "role": "user"},
{"user_email": "user2@example.com", "role": "admin"},
Member(user_email="user1@example.com", role="user"),
Member(user_email="user2@example.com", role="admin"),
]
request_data = TeamMemberAddRequest(
team_id="test-team-123",
@ -1505,9 +1505,9 @@ async def test_list_team_v2_security_check_non_admin_user():
assert exc_info.value.status_code == 401
assert "Only admin users can query all teams/other teams" in str(
exc_info.value.detail["error"]
exc_info.value.detail
)
assert LitellmUserRoles.INTERNAL_USER.value in str(exc_info.value.detail["error"])
assert LitellmUserRoles.INTERNAL_USER.value in str(exc_info.value.detail)
@pytest.mark.asyncio
@ -1545,7 +1545,7 @@ async def test_list_team_v2_security_check_non_admin_user_other_user():
assert exc_info.value.status_code == 401
assert "Only admin users can query all teams/other teams" in str(
exc_info.value.detail["error"]
exc_info.value.detail
)
@ -1654,3 +1654,55 @@ async def test_list_team_v2_security_check_admin_user():
assert "teams" in result
assert "total" in result
assert result["total"] == 2
@pytest.mark.asyncio
async def test_team_member_delete_cleans_membership(mock_db_client, mock_admin_auth):
"""
Verify that /team/member_delete removes the corresponding LiteLLM_TeamMembership row
so the same user can be re-added without unique constraint issues.
"""
from litellm.proxy._types import TeamMemberDeleteRequest
from litellm.proxy.management_endpoints.team_endpoints import team_member_delete
test_team_id = "team-del-123"
test_user_id = "user@example.com"
# Mock Team row with the user as a member
mock_team_row = MagicMock()
mock_team_row.model_dump.return_value = {
"team_id": test_team_id,
"members_with_roles": [
{"user_id": test_user_id, "user_email": None, "role": "user"}
],
"team_member_permissions": [],
"metadata": {},
"models": [],
"spend": 0.0,
}
# Configure DB mocks used by team_member_delete
mock_db_client.db.litellm_teamtable.find_unique = AsyncMock(return_value=mock_team_row)
mock_db_client.db.litellm_teamtable.update = AsyncMock(return_value=mock_team_row)
# User row to allow removal from user's teams list
mock_user_row = MagicMock()
mock_user_row.user_id = test_user_id
mock_user_row.teams = [test_team_id]
mock_db_client.db.litellm_usertable.find_many = AsyncMock(return_value=[mock_user_row])
mock_db_client.db.litellm_usertable.update = AsyncMock(return_value=MagicMock())
# Membership deletion should be called
mock_db_client.db.litellm_teammembership = MagicMock()
mock_db_client.db.litellm_teammembership.delete_many = AsyncMock(return_value=MagicMock())
# Execute
await team_member_delete(
data=TeamMemberDeleteRequest(team_id=test_team_id, user_id=test_user_id),
user_api_key_dict=mock_admin_auth,
)
# Assert membership cleanup executed
mock_db_client.db.litellm_teammembership.delete_many.assert_awaited_with(
where={"team_id": test_team_id, "user_id": test_user_id}
)

View file

@ -253,3 +253,34 @@ class TestReasoningContentFinalResponse:
]
assert len(reasoning_items) == 1, "Should have exactly one reasoning item"
assert reasoning_items[0].content[0].text == "Reasoning for first answer"
def test_streaming_chunk_id_raw():
"""Test that streaming chunk IDs are raw (not encoded) to match OpenAI format"""
chunk = ModelResponseStream(
id="chunk-123",
created=1234567890,
model="test-model",
object="chat.completion.chunk",
choices=[
StreamingChoices(
finish_reason=None,
index=0,
delta=Delta(content="Hello", role="assistant"),
)
],
)
iterator = LiteLLMCompletionStreamingIterator(
litellm_custom_stream_wrapper=AsyncMock(),
request_input="Test input",
responses_api_request={},
custom_llm_provider="openai",
litellm_metadata={"model_info": {"id": "gpt-4"}},
)
result = iterator._transform_chat_completion_chunk_to_response_api_chunk(chunk)
# Streaming chunk IDs should be raw (like OpenAI's msg_xxx format)
assert result.item_id == "chunk-123" # Should be raw, not encoded
assert not result.item_id.startswith("resp_") # Should NOT have resp_ prefix

View file

@ -844,6 +844,7 @@ async def test_supports_tool_choice():
or "o1" in model_name
or "o3" in model_name
or "mistral" in model_name
or "oci" in model_name
):
continue
@ -2317,8 +2318,9 @@ def test_block_key_hashing_logic():
Test that block_key() function only hashes keys that start with "sk-"
"""
import hashlib
from litellm.proxy.utils import hash_token
# Test cases: (input_key, should_be_hashed, expected_output)
test_cases = [
("sk-1234567890abcdef", True, hash_token("sk-1234567890abcdef")),
@ -2394,7 +2396,7 @@ def test_generate_gcp_iam_access_token_import_error():
"""
# Import the function first, before mocking
from litellm._redis import _generate_gcp_iam_access_token
# Mock the import to fail when the function tries to import google.cloud.iam_credentials_v1
original_import = __builtins__['__import__']

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[19813,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-9e1803a09e9ae8da.js","498","static/chunks/498-ee02f9b58491d7a9.js","154","static/chunks/154-78c3416dcb61977f.js","162","static/chunks/162-8529572226f208c5.js","172","static/chunks/172-08ae62d50ce1f0e7.js","931","static/chunks/app/page-0a9a9f137522a76c.js"],"default",1]
3:I[6691,["665","static/chunks/3014691f-b7b79b78e27792f3.js","990","static/chunks/13b76428-ebdf3012af0e4489.js","416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-3523e0e07cf314f6.js","683","static/chunks/683-07087d813e7eeb43.js","154","static/chunks/154-66d79df6143c694f.js","162","static/chunks/162-9e6f5133e328d61f.js","172","static/chunks/172-1c7afccd96ceca39.js","931","static/chunks/app/page-1d51309983956823.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["__PAGE__",{}]},"$undefined","$undefined",true],["",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[52829,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","154","static/chunks/154-78c3416dcb61977f.js","162","static/chunks/162-8529572226f208c5.js","418","static/chunks/app/model_hub/page-b26e0d313b582dbf.js"],"default",1]
3:I[52829,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","154","static/chunks/154-66d79df6143c694f.js","162","static/chunks/162-9e6f5133e328d61f.js","418","static/chunks/app/model_hub/page-b26e0d313b582dbf.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["model_hub",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[22775,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-9e1803a09e9ae8da.js","154","static/chunks/154-78c3416dcb61977f.js","162","static/chunks/162-8529572226f208c5.js","172","static/chunks/172-08ae62d50ce1f0e7.js","25","static/chunks/app/model_hub_table/page-d080c5775ebaf3a1.js"],"default",1]
3:I[22775,["416","static/chunks/416-ad6bd55a20a586bd.js","90","static/chunks/90-d2b5ed6f7f6e342e.js","866","static/chunks/866-3523e0e07cf314f6.js","154","static/chunks/154-66d79df6143c694f.js","162","static/chunks/162-9e6f5133e328d61f.js","172","static/chunks/172-1c7afccd96ceca39.js","25","static/chunks/app/model_hub_table/page-d080c5775ebaf3a1.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["model_hub_table",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["model_hub_table",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","model_hub_table","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

File diff suppressed because one or more lines are too long

View file

@ -1,7 +1,7 @@
2:I[19107,[],"ClientPageRoot"]
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","416","static/chunks/416-ad6bd55a20a586bd.js","154","static/chunks/154-78c3416dcb61977f.js","461","static/chunks/app/onboarding/page-883c32e6b072b842.js"],"default",1]
3:I[12011,["665","static/chunks/3014691f-b7b79b78e27792f3.js","416","static/chunks/416-ad6bd55a20a586bd.js","154","static/chunks/154-66d79df6143c694f.js","461","static/chunks/app/onboarding/page-7e4cd2bb92dbf9ce.js"],"default",1]
4:I[4707,[],""]
5:I[36423,[],""]
0:["rkJYRcYqQ8OBbL83KQCc4",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/fe37e928fd602d9e.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
0:["poVnZDt3J0aYERGVwpJKO",[[["",{"children":["onboarding",{"children":["__PAGE__",{}]}]},"$undefined","$undefined",true],["",{"children":["onboarding",{"children":["__PAGE__",{},[["$L1",["$","$L2",null,{"props":{"params":{},"searchParams":{}},"Component":"$3"}],null],null],null]},[null,["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children","onboarding","children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":"$undefined","notFoundStyles":"$undefined"}]],null]},[[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/31b7f215e119031e.css","precedence":"next","crossOrigin":"$undefined"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/css/0dbff0867726409d.css","precedence":"next","crossOrigin":"$undefined"}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"__className_b0dd8a","children":["$","$L4",null,{"parallelRouterKey":"children","segmentPath":["children"],"error":"$undefined","errorStyles":"$undefined","errorScripts":"$undefined","template":["$","$L5",null,{}],"templateStyles":"$undefined","templateScripts":"$undefined","notFound":[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":"404"}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],"notFoundStyles":[]}]}]}]],null],null],["$L6",null]]]]
6:[["$","meta","0",{"name":"viewport","content":"width=device-width, initial-scale=1"}],["$","meta","1",{"charSet":"utf-8"}],["$","title","2",{"children":"LiteLLM Dashboard"}],["$","meta","3",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","4",{"rel":"icon","href":"/favicon.ico","type":"image/x-icon","sizes":"16x16"}],["$","link","5",{"rel":"icon","href":"./favicon.ico"}],["$","meta","6",{"name":"next-size-adjust"}]]
1:null

View file

@ -1,12 +1,11 @@
import React, { useEffect, useState } from "react";
import { Form, Table, Input } from "antd";
import { Text, TextInput } from "@tremor/react";
import { Row, Col } from "antd";
import { Form, Table } from "antd";
import { TextInput } from "@tremor/react";
import { Tooltip } from "../atoms/index";
const ConditionalPublicModelName: React.FC = () => {
// Access the form instance
const form = Form.useFormInstance();
const [tableKey, setTableKey] = useState(0); // Add a key to force table re-render
const [tableKey, setTableKey] = useState(0);// Add a key to force table re-render
// Watch the 'model' field for changes and ensure it's always an array
const modelValue = Form.useWatch('model', form) || [];
@ -14,7 +13,6 @@ const ConditionalPublicModelName: React.FC = () => {
const customModelName = Form.useWatch('custom_model_name', form);
const showPublicModelName = !selectedModels.includes('all-wildcard');
// Force table to re-render when custom model name changes
useEffect(() => {
if (customModelName && selectedModels.includes('custom')) {
@ -71,9 +69,39 @@ const ConditionalPublicModelName: React.FC = () => {
if (!showPublicModelName) return null;
const publicNameTooltipContent = (
<>
<div className="mb-2 font-normal">
The name you specify in your API calls to LiteLLM Proxy
</div>
<div className="mb-2 font-normal">
<strong>Example:</strong> If you name your public model <code className="bg-gray-700 px-1 py-0.5 rounded text-xs">example-name</code>
, and choose <code className="bg-gray-700 px-1 py-0.5 rounded text-xs">openai/qwen-plus-latest</code> as the LiteLLM model
</div>
<div className="mb-2 font-normal">
<strong>Usage:</strong> You make an API call to the LiteLLM proxy with <code className="bg-gray-700 px-1 py-0.5 rounded text-xs">model = &quot;example-name&quot;</code>
</div>
<div className="font-normal">
<strong>Result:</strong> LiteLLM sends <code className="bg-gray-700 px-1 py-0.5 rounded text-xs">qwen-plus-latest</code> to the provider
</div>
</>
);
const liteLLMModelTooltipContent = (
<div>The model name LiteLLM will send to the LLM API</div>
);
const columns = [
{
title: 'Public Name',
title: (
<span className="flex items-center">
Public Model Name
<Tooltip
content={publicNameTooltipContent}
width="500px"
/>
</span>
),
dataIndex: 'public_name',
key: 'public_name',
render: (text: string, record: any, index: number) => {
@ -90,7 +118,15 @@ const ConditionalPublicModelName: React.FC = () => {
}
},
{
title: 'LiteLLM Model',
title: (
<span className="flex items-center">
LiteLLM Model Name
<Tooltip
content={liteLLMModelTooltipContent}
width="360px"
/>
</span>
),
dataIndex: 'litellm_model',
key: 'litellm_model',
}

View file

@ -70,7 +70,7 @@ const LiteLLMModelNameField: React.FC<LiteLLMModelNameFieldProps> = ({
<>
<Form.Item
label="LiteLLM Model Name(s)"
tooltip="Actual model name used for making litellm.completion() / litellm.embedding() call."
tooltip="The model name LiteLLM will send to the LLM API"
className="mb-0"
>
<Form.Item
@ -145,9 +145,9 @@ const LiteLLMModelNameField: React.FC<LiteLLMModelNameFieldProps> = ({
</Form.Item>
<Row>
<Col span={10}></Col>
<Col span={10}>
<Col span={14}>
<Text className="mb-3 mt-1">
Actual model name used for making litellm.completion() call. We loadbalance models with the same public name
The model name LiteLLM will send to the LLM API
</Text>
</Col>
</Row>

View file

@ -0,0 +1,70 @@
import React, { useState, useRef } from "react"
import { QuestionCircleOutlined } from "@ant-design/icons"
interface TooltipProps {
content: React.ReactNode
children?: React.ReactNode
width?: string
className?: string
}
export const Tooltip: React.FC<TooltipProps> = ({ content, children, width = "auto", className = "" }) => {
const [showTooltip, setShowTooltip] = useState(false)
const [tooltipPosition, setTooltipPosition] = useState<"top" | "bottom">("top")
const tooltipRef = useRef<HTMLDivElement>(null)
// Function to check if tooltip would fit above
const checkTooltipPosition = () => {
if (tooltipRef.current) {
const rect = tooltipRef.current.getBoundingClientRect()
const tooltipHeight = 300 // Approximate height of the tooltip
const spaceAbove = rect.top
const spaceBelow = window.innerHeight - rect.bottom
if (spaceAbove < tooltipHeight && spaceBelow > tooltipHeight) {
setTooltipPosition("bottom")
} else {
setTooltipPosition("top")
}
}
}
return (
<div className="relative inline-block" ref={tooltipRef}>
{children || (
<QuestionCircleOutlined
className="ml-1 text-gray-500 cursor-help"
onMouseEnter={() => {
checkTooltipPosition()
setShowTooltip(true)
}}
onMouseLeave={() => setShowTooltip(false)}
/>
)}
{showTooltip && (
<div
className={`absolute left-1/2 -translate-x-1/2 z-50 bg-black/90 text-white p-2 rounded-md text-sm font-normal shadow-lg ${className}`}
style={{
[tooltipPosition === "top" ? "bottom" : "top"]: "100%",
width: width,
marginBottom: tooltipPosition === "top" ? "8px" : "0",
marginTop: tooltipPosition === "bottom" ? "8px" : "0",
}}
>
{content}
<div
className="absolute left-1/2 -translate-x-1/2 w-0 h-0"
style={{
top: tooltipPosition === "top" ? "100%" : "auto",
bottom: tooltipPosition === "bottom" ? "100%" : "auto",
borderTop: tooltipPosition === "top" ? "6px solid rgba(0, 0, 0, 0.9)" : "6px solid transparent",
borderBottom: tooltipPosition === "bottom" ? "6px solid rgba(0, 0, 0, 0.9)" : "6px solid transparent",
borderLeft: "6px solid transparent",
borderRight: "6px solid transparent",
}}
/>
</div>
)}
</div>
)
}

View file

@ -0,0 +1 @@
export { Tooltip } from './Tooltip';

View file

@ -1,6 +1,7 @@
import React, { useState, useEffect } from "react";
import { Card, Text, Grid, Button } from "@tremor/react";
import { Typography, message, Divider, Spin, Checkbox } from "antd";
import { Typography, Divider, Spin, Checkbox } from "antd";
import NotificationsManager from "../molecules/notifications_manager";
import { getEmailEventSettings, updateEmailEventSettings, resetEmailEventSettings } from "../networking";
import { EmailEvent } from "../../types";
import { EmailEventSetting } from "./types";
@ -31,7 +32,7 @@ const EmailEventSettings: React.FC<EmailEventSettingsProps> = ({
setEventSettings(response.settings);
} catch (error) {
console.error("Failed to fetch email event settings:", error);
message.error("Failed to fetch email event settings");
NotificationsManager.fromBackend(error);
} finally {
setLoading(false);
}
@ -49,10 +50,10 @@ const EmailEventSettings: React.FC<EmailEventSettingsProps> = ({
try {
await updateEmailEventSettings(accessToken, { settings: eventSettings });
message.success("Email event settings updated successfully");
NotificationsManager.success("Email event settings updated successfully");
} catch (error) {
console.error("Failed to update email event settings:", error);
message.error("Failed to update email event settings");
NotificationsManager.fromBackend(error);
}
};
@ -61,12 +62,12 @@ const EmailEventSettings: React.FC<EmailEventSettingsProps> = ({
try {
await resetEmailEventSettings(accessToken);
message.success("Email event settings reset to defaults");
NotificationsManager.success("Email event settings reset to defaults");
// Refresh settings after reset
fetchEventSettings();
} catch (error) {
console.error("Failed to reset email event settings:", error);
message.error("Failed to reset email event settings");
NotificationsManager.fromBackend(error);
}
};

View file

@ -7,7 +7,8 @@ import {
TextInput,
TableCell,
} from "@tremor/react";
import { Typography, message, Divider } from "antd";
import { Typography } from "antd";
import NotificationsManager from "./molecules/notifications_manager";
import { serviceHealthCheck, setCallbacksCall } from "./networking";
import { EmailEventSettings } from "./email_events";
@ -24,7 +25,7 @@ const EmailSettings: React.FC<EmailSettingsProps> = ({
premiumUser,
alerts,
}) => {
const handleSaveEmailSettings = () => {
const handleSaveEmailSettings = async () => {
if (!accessToken) {
return;
}
@ -52,12 +53,11 @@ const EmailSettings: React.FC<EmailSettingsProps> = ({
environment_variables: updatedVariables,
};
try {
setCallbacksCall(accessToken, payload);
await setCallbacksCall(accessToken, payload);
NotificationsManager.success("Email settings updated successfully");
} catch (error) {
message.error("Failed to update alerts: " + error, 20);
NotificationsManager.fromBackend(error);
}
message.success("Email settings updated successfully");
}
return (
@ -191,9 +191,15 @@ const EmailSettings: React.FC<EmailSettingsProps> = ({
Save Changes
</Button>
<Button
onClick={() =>
accessToken && serviceHealthCheck(accessToken, "email")
}
onClick={async () => {
if (!accessToken) return;
try {
await serviceHealthCheck(accessToken, "email");
NotificationsManager.success("Email test triggered. Check your configured email inbox/logs.");
} catch (error) {
NotificationsManager.fromBackend(error);
}
}}
className="mx-2"
>
Test Email Alerts

View file

@ -0,0 +1,318 @@
import { notification } from "antd"
import { parseErrorMessage } from "../shared/errorUtils"
type Placement = "top" | "topLeft" | "topRight" | "bottom" | "bottomLeft" | "bottomRight"
type NotificationConfig = {
message?: string
description?: string
duration?: number
placement?: Placement
key?: string
}
type NotificationConfigResolved = Omit<NotificationConfig, "message"> & { message: string }
function defaultPlacement(): Placement {
return "topRight"
}
function normalize(input: string | NotificationConfig, fallbackTitle: string): NotificationConfigResolved {
if (typeof input === "string") return { message: fallbackTitle, description: input }
return { message: input.message ?? fallbackTitle, ...input }
}
function toIntMaybe(val: any): number | undefined {
if (typeof val === "number") return val
if (typeof val === "string" && /^\d+$/.test(val)) return parseInt(val, 10)
return undefined
}
const AUTH_MATCH = [
"invalid api key",
"invalid authorization header format",
"authentication error",
"invalid proxy server token",
"invalid jwt token",
"invalid jwt submitted",
"unauthorized access to metrics endpoint",
];
const FORBIDDEN_MATCH = [
"admin-only endpoint",
"not allowed to access model",
"user does not have permission",
"access forbidden",
"invalid credentials used to access ui",
"user not allowed to access proxy",
];
const DB_MATCH = [
"db not connected",
"database not initialized",
"no db connected",
"prisma client not initialized",
"service unhealthy",
];
const ROUTER_MATCH = [
"no models configured on proxy",
"llm router not initialized",
"no deployments available",
"no healthy deployment available",
"not allowed to access model due to tags configuration",
"invalid model name passed in",
];
const RATE_LIMIT_EXTRA = [
"deployment over user-defined ratelimit",
"crossed tpm / rpm / max parallel request limit",
"max parallel request limit",
];
const BUDGET_MATCH = [
"budget exceeded",
"crossed budget",
"provider budget",
];
const ENTERPRISE_MATCH = [
"must be a litellm enterprise user",
"only be available for liteLLM enterprise users",
"missing litellm-enterprise package",
"only available on the docker image",
"enterprise feature",
"premium user",
];
const VALIDATION_MATCH = [
"invalid json payload",
"invalid request type",
"invalid key format",
"invalid hash key",
"invalid sort column",
"invalid sort order",
"invalid limit",
"invalid file type",
"invalid field",
"invalid date format",
];
const NOT_FOUND_MATCH = [
"model not found",
"model with id",
"credential not found",
"user not found",
"team not found",
"organization not found",
"mcp server with id",
"tool '", // will combine with “not found” in message
];
const EXISTS_MATCH = [
"already exists",
"team member is already in team",
"user already exists",
];
const GUARDRAIL_MATCH = [
"violated openai moderation policy",
"violated jailbreak threshold",
"violated prompt_injection threshold",
"violated content safety policy",
"violated lasso guardrail policy",
"blocked by pillar security guardrail",
"violated azure prompt shield guardrail policy",
"content blocked by model armor",
"response blocked by model armor",
"streaming response blocked by model armor",
"guardrail",
"moderation",
];
const FILE_UPLOAD_MATCH = [
"invalid purpose",
"service must be specified",
"invalid response - response.response is none",
];
const CLOUDZERO_MATCH = [
"cloudzero settings not configured",
"failed to decrypt cloudzero api key",
"cloudzero settings not found",
];
function titleFor(status?: number, desc?: string): string {
const d = (desc || "").toLowerCase();
if (AUTH_MATCH.some(s => d.includes(s))) return "Authentication Error";
if (FORBIDDEN_MATCH.some(s => d.includes(s))) return "Access Denied";
if (DB_MATCH?.some?.((s:string)=>d.includes(s)) || status === 503) return "Service Unavailable";
if (BUDGET_MATCH?.some?.((s:string)=>d.includes(s))) return "Budget Exceeded";
if (ENTERPRISE_MATCH?.some?.((s:string)=>d.includes(s))) return "Feature Unavailable";
if (ROUTER_MATCH?.some?.((s:string)=>d.includes(s))) return "Routing Error";
if (EXISTS_MATCH.some(s => d.includes(s))) return "Already Exists";
if (GUARDRAIL_MATCH.some(s => d.includes(s))) return "Content Blocked";
if (FILE_UPLOAD_MATCH.some(s => d.includes(s))) return "Validation Error";
if (CLOUDZERO_MATCH.some(s => d.includes(s))) return "Integration Error";
if (VALIDATION_MATCH.some(s => d.includes(s))) return "Validation Error";
if (status === 404 || d.includes("not found") || NOT_FOUND_MATCH.some(s => d.includes(s))) return "Not Found";
if (status === 429 || d.includes("rate limit") || d.includes("tpm") || d.includes("rpm") || RATE_LIMIT_EXTRA?.some?.((s:string)=>d.includes(s))) return "Rate Limit Exceeded";
if (status && status >= 500) return "Server Error";
if (status === 401) return "Authentication Error";
if (status === 403) return "Access Denied";
if (d.includes("enterprise") || d.includes("premium")) return "Info";
if (status && status >= 400) return "Request Error";
return "Error";
}
const SUCCESS_MATCH = [
"created successfully",
"updated successfully",
"deleted successfully",
"credential created successfully",
"model added successfully",
"team created successfully",
"user created successfully",
"organization created successfully",
"cloudzero settings initialized successfully",
"cloudzero settings updated successfully",
"cloudzero export completed successfully",
"mock llm request made",
"mock slack alert sent",
"mock email alert sent",
"spend for all api keys and teams reset successfully",
"monthlyglobalspend view refreshed",
"cache cleared successfully",
"cache set successfully",
"ip ",
"deleted successfully"
];
const INFO_MATCH = [
"rate limit reached for deployment",
"deployment cooldown period active",
];
const DEPRECATION_FEATURE_WARN_MATCH = [
"this feature is only available for litellm enterprise users",
"enterprise features are not available",
"regenerating virtual keys is an enterprise feature",
"trying to set allowed_routes. this is an enterprise feature",
];
const CONFIG_WARN_MATCH = [
"invalid maximum_spend_logs_retention_interval value",
"error has invalid or non-convertible code",
"failed to save health check to database",
];
function classifyGeneralMessage(desc?: string): { kind: "success" | "info" | "warning"; title: string } | null {
const d = (desc || "").toLowerCase();
if (SUCCESS_MATCH.some(s => d.includes(s))) return { kind: "success", title: "Success" };
if (DEPRECATION_FEATURE_WARN_MATCH.some(s => d.includes(s))) return { kind: "warning", title: "Feature Notice" };
if (CONFIG_WARN_MATCH.some(s => d.includes(s))) return { kind: "warning", title: "Configuration Warning" };
if (INFO_MATCH.some(s => d.includes(s))) return { kind: "warning", title: "Rate Limit" }; // show as warning for visibility
return null;
}
function extractStatus(input: any): number | undefined {
return toIntMaybe(input?.response?.status) ?? toIntMaybe(input?.status_code) ?? toIntMaybe(input?.code);
}
function extractDescription(input: any): string {
if (typeof input === "string") return input; // raw error string
const backendMsg =
input?.response?.data?.error?.message ??
input?.response?.data?.message ??
input?.response?.data?.error ??
input?.detail ??
input?.message ??
input;
return parseErrorMessage(backendMsg);
}
function looksErrorPayload(input: any, status?: number): boolean {
if (status !== undefined) return true;
if (input instanceof Error) return true;
if (typeof input === "string") return true; // treat raw strings passed to fromBackend as errors
if (input && typeof input === "object" && ("error" in input || "detail" in input)) return true;
return false;
}
const NotificationManager = {
error(input: string | NotificationConfig) {
const cfg = normalize(input, "Error")
notification.error({
...cfg,
placement: cfg.placement ?? defaultPlacement(),
duration: cfg.duration ?? 6,
})
},
warning(input: string | NotificationConfig) {
const cfg = normalize(input, "Warning")
notification.warning({
...cfg,
placement: cfg.placement ?? defaultPlacement(),
duration: cfg.duration ?? 5,
})
},
info(input: string | NotificationConfig) {
const cfg = normalize(input, "Info")
notification.info({
...cfg,
placement: cfg.placement ?? defaultPlacement(),
duration: cfg.duration ?? 4,
})
},
success(input: string | NotificationConfig) {
const cfg = normalize(input, "Success")
notification.success({
...cfg,
placement: cfg.placement ?? defaultPlacement(),
duration: cfg.duration ?? 3.5,
})
},
fromBackend(input: any, extra?: Omit<NotificationConfig, "message" | "description">) {
const status = extractStatus(input);
const description = extractDescription(input);
const base = { ...(extra ?? {}), description, placement: extra?.placement ?? defaultPlacement() };
if (looksErrorPayload(input, status)) {
const title = titleFor(status, description);
const payload = { ...base, message: title };
if (title === "Rate Limit Exceeded" || title === "Info" || title === "Budget Exceeded" || title === "Feature Unavailable" || title === "Content Blocked" || title === "Integration Error") {
notification.warning({ ...payload, duration: extra?.duration ?? 7 }); return;
}
if (title === "Server Error") { notification.error({ ...payload, duration: extra?.duration ?? 8 }); return; }
if (title === "Request Error" || title === "Authentication Error" || title === "Access Denied" || title === "Not Found" || title === "Error") {
notification.error({ ...payload, duration: extra?.duration ?? 6 }); return;
}
notification.info({ ...payload, duration: extra?.duration ?? 4 }); return;
}
// Non-error: success/info/warning classifier
const cls = classifyGeneralMessage(description);
const payload = { ...base, message: cls?.title ?? "Info" };
if (cls?.kind === "success") { notification.success({ ...payload, duration: extra?.duration ?? 3.5 }); return; }
if (cls?.kind === "warning") { notification.warning({ ...payload, duration: extra?.duration ?? 6 }); return; }
notification.info({ ...payload, duration: extra?.duration ?? 4 });
},
clear() {
notification.destroy()
},
}
export default NotificationManager

View file

@ -1,3 +1,10 @@
// Shared date formatter for daily activity endpoints
export const formatDate = (date: Date) => {
const year = date.getFullYear();
const month = String(date.getMonth() + 1).padStart(2, '0');
const day = String(date.getDate()).padStart(2, '0');
return `${year}-${month}-${day}`;
};
/**
* Helper file for calls being made to proxy
*/
@ -1457,13 +1464,6 @@ export const userDailyActivityCall = async (
? `${proxyBaseUrl}/user/daily/activity`
: `/user/daily/activity`;
const queryParams = new URLSearchParams();
// Format dates as YYYY-MM-DD for the API
const formatDate = (date: Date) => {
const year = date.getFullYear();
const month = String(date.getMonth() + 1).padStart(2, '0');
const day = String(date.getDate()).padStart(2, '0');
return `${year}-${month}-${day}`;
};
queryParams.append("start_date", formatDate(startTime));
queryParams.append("end_date", formatDate(endTime));
queryParams.append("page_size", "1000");
@ -1510,13 +1510,6 @@ export const tagDailyActivityCall = async (
? `${proxyBaseUrl}/tag/daily/activity`
: `/tag/daily/activity`;
const queryParams = new URLSearchParams();
// Format dates as YYYY-MM-DD for the API
const formatDate = (date: Date) => {
const year = date.getFullYear();
const month = String(date.getMonth() + 1).padStart(2, '0');
const day = String(date.getDate()).padStart(2, '0');
return `${year}-${month}-${day}`;
};
queryParams.append("start_date", formatDate(startTime));
queryParams.append("end_date", formatDate(endTime));
queryParams.append("page_size", "1000");
@ -1566,13 +1559,6 @@ export const teamDailyActivityCall = async (
? `${proxyBaseUrl}/team/daily/activity`
: `/team/daily/activity`;
const queryParams = new URLSearchParams();
// Format dates as YYYY-MM-DD for the API
const formatDate = (date: Date) => {
const year = date.getFullYear();
const month = String(date.getMonth() + 1).padStart(2, '0');
const day = String(date.getDate()).padStart(2, '0');
return `${year}-${month}-${day}`;
};
queryParams.append("start_date", formatDate(startTime));
queryParams.append("end_date", formatDate(endTime));
queryParams.append("page_size", "1000");
@ -3198,6 +3184,55 @@ export interface User {
[key: string]: string; // Include any other potential keys in the dictionary
}
export const userDailyActivityAggregatedCall = async (
accessToken: String,
startTime: Date,
endTime: Date
) => {
/**
* Get aggregated daily user activity (no pagination)
*/
try {
let url = proxyBaseUrl
? `${proxyBaseUrl}/user/daily/activity/aggregated`
: `/user/daily/activity/aggregated`;
const queryParams = new URLSearchParams();
// Format dates as YYYY-MM-DD for the API
const formatDate = (date: Date) => {
const year = date.getFullYear();
const month = String(date.getMonth() + 1).padStart(2, '0');
const day = String(date.getDate()).padStart(2, '0');
return `${year}-${month}-${day}`;
};
queryParams.append("start_date", formatDate(startTime));
queryParams.append("end_date", formatDate(endTime));
const queryString = queryParams.toString();
if (queryString) {
url += `?${queryString}`;
}
const response = await fetch(url, {
method: "GET",
headers: {
[globalLitellmHeaderName]: `Bearer ${accessToken}`,
"Content-Type": "application/json",
},
});
if (!response.ok) {
const errorData = await response.text();
handleError(errorData);
throw new Error("Network response was not ok");
}
const data = await response.json();
return data;
} catch (error) {
console.error("Failed to fetch aggregated user daily activity:", error);
throw error;
}
};
export const userGetAllUsersCall = async (
accessToken: String,
role: String
@ -4229,9 +4264,6 @@ export const serviceHealthCheck = async (
}
const data = await response.json();
message.success(
`Test request to ${service} made - check logs/alerts on ${service} to verify`
);
// You can add additional logic here based on the response if needed
return data;
} catch (error) {

View file

@ -34,7 +34,7 @@ import {
import AdvancedDatePicker from "./shared/advanced_date_picker"
import { AreaChart } from "@tremor/react"
import { userDailyActivityCall, tagListCall } from "./networking"
import { userDailyActivityCall, userDailyActivityAggregatedCall, tagListCall } from "./networking"
import { Tag } from "./tag_management/types"
import ViewUserSpend from "./view_user_spend"
import TopKeyView from "./top_key_view"
@ -304,16 +304,22 @@ const NewUsagePage: React.FC<NewUsagePageProps> = ({ accessToken, userRole, user
const endTime = new Date(dateValue.to)
try {
// Get first page
// Prefer aggregated endpoint to avoid many page requests
try {
const aggregated = await userDailyActivityAggregatedCall(accessToken, startTime, endTime)
setUserSpendData(aggregated)
return
} catch (e) {
// Fallback to paginated calls if aggregated endpoint is unavailable
}
const firstPageData = await userDailyActivityCall(accessToken, startTime, endTime)
// If only one page, just set the data
if (firstPageData.metadata.total_pages <= 1) {
setUserSpendData(firstPageData)
return
}
// Fetch all pages
const allResults = [...firstPageData.results]
const aggregatedMetadata = { ...firstPageData.metadata }
@ -329,7 +335,6 @@ const NewUsagePage: React.FC<NewUsagePageProps> = ({ accessToken, userRole, user
}
}
// Combine all results with the first page's metadata
setUserSpendData({
results: allResults,
metadata: aggregatedMetadata,

View file

@ -30,8 +30,8 @@ import {
Input,
Select,
Button as Button2,
message,
} from "antd";
import NotificationsManager from "./molecules/notifications_manager";
import EmailSettings from "./email_settings";
const { Title, Paragraph } = Typography;
@ -49,6 +49,7 @@ import {
callbackInfo,
Callbacks,
} from "./callback_info_helpers";
import { parseErrorMessage } from "./shared/errorUtils";
interface SettingsPageProps {
accessToken: string | null;
userRole: string | null;
@ -84,7 +85,8 @@ const Settings: React.FC<SettingsPageProps> = ({
const [callbacks, setCallbacks] = useState<AlertingObject[]>([]);
const [alerts, setAlerts] = useState<any[]>([]);
const [isModalVisible, setIsModalVisible] = useState(false);
const [form] = Form.useForm();
const [addForm] = Form.useForm();
const [editForm] = Form.useForm();
const [selectedCallback, setSelectedCallback] = useState<string | null>(null);
const [catchAllWebhookURL, setCatchAllWebhookURL] = useState<string>("");
const [alertToWebhooks, setAlertToWebhooks] = useState<
@ -106,6 +108,15 @@ const Settings: React.FC<SettingsPageProps> = ({
const [showDeleteConfirmModal, setShowDeleteConfirmModal] = useState(false);
const [callbackToDelete, setCallbackToDelete] = useState<string | null>(null);
useEffect(() => {
if (showEditCallback && selectedEditCallback) {
const normalized = Object.fromEntries(
Object.entries(selectedEditCallback.variables || {}).map(([k, v]) => [k, v ?? ""])
);
editForm.setFieldsValue(normalized)
}
}, [showEditCallback, selectedEditCallback, editForm]);
const handleSwitchChange = (alertName: string) => {
if (activeAlerts.includes(alertName)) {
setActiveAlerts(activeAlerts.filter((alert) => alert !== alertName));
@ -155,7 +166,7 @@ const Settings: React.FC<SettingsPageProps> = ({
};
const updateCallbackCall = async (formValues: Record<string, any>) => {
if (!accessToken) {
if (!accessToken || !selectedEditCallback) {
return;
}
@ -167,17 +178,27 @@ const Settings: React.FC<SettingsPageProps> = ({
}
});
let payload = {
environment_variables: env_vars,
};
environment_variables: formValues,
litellm_settings: {
"success_callback": [selectedEditCallback.name]
}
}
try {
await setCallbacksCall(accessToken, payload);
message.success(`Callback added successfully`);
setIsModalVisible(false);
form.resetFields();
setSelectedCallback(null);
NotificationsManager.success("Callback updated successfully");
setShowEditCallback(false);
editForm.resetFields();
setSelectedEditCallback(null);
// Refresh the callbacks list
if (userID && userRole) {
const updatedData = await getCallbacksCall(accessToken, userID, userRole);
setCallbacks(updatedData.callbacks);
}
} catch (error) {
message.error("Failed to add callback: " + error, 20);
NotificationsManager.fromBackend(error);
}
};
@ -196,7 +217,7 @@ const Settings: React.FC<SettingsPageProps> = ({
});
let payload = {
environment_variables: env_vars,
environment_variables: formValues,
litellm_settings: {
success_callback: [new_callback],
},
@ -204,12 +225,17 @@ const Settings: React.FC<SettingsPageProps> = ({
try {
await setCallbacksCall(accessToken, payload);
message.success(`Callback ${new_callback} added successfully`);
setIsModalVisible(false);
form.resetFields();
NotificationsManager.success(`Callback ${new_callback} added successfully`);
setShowAddCallbacksModal(false);
addForm.resetFields();
setSelectedCallback(null);
setSelectedCallbackParams([]);
// Refresh the callbacks list
const updatedData = await getCallbacksCall(accessToken, userID || "", userRole || "");
setCallbacks(updatedData.callbacks);
} catch (error) {
message.error("Failed to add callback: " + error, 20);
NotificationsManager.fromBackend(error);
}
};
@ -225,7 +251,7 @@ const Settings: React.FC<SettingsPageProps> = ({
}
};
const handleSaveAlerts = () => {
const handleSaveAlerts = async () => {
if (!accessToken) {
return;
}
@ -247,12 +273,11 @@ const Settings: React.FC<SettingsPageProps> = ({
};
try {
setCallbacksCall(accessToken, payload);
await setCallbacksCall(accessToken, payload);
} catch (error) {
message.error("Failed to update alerts: " + error, 20);
NotificationsManager.fromBackend(error);
}
message.success("Alerts updated successfully");
NotificationsManager.success("Alerts updated successfully");
};
const handleSaveChanges = (callback: any) => {
if (!accessToken) {
@ -277,10 +302,9 @@ const Settings: React.FC<SettingsPageProps> = ({
try {
setCallbacksCall(accessToken, payload);
} catch (error) {
message.error("Failed to update callback: " + error, 20);
NotificationsManager.fromBackend(error);
}
message.success("Callback updated successfully");
NotificationsManager.success("Callback updated successfully");
};
const handleOk = () => {
@ -288,7 +312,7 @@ const Settings: React.FC<SettingsPageProps> = ({
return;
}
// Handle form submission
form.validateFields().then((values) => {
addForm.validateFields().then((values) => {
// Call API to add the callback
let payload;
if (values.callback === "langfuse") {
@ -365,7 +389,7 @@ const Settings: React.FC<SettingsPageProps> = ({
};
}
setIsModalVisible(false);
form.resetFields();
addForm.resetFields();
setSelectedCallback(null);
});
};
@ -382,7 +406,7 @@ const Settings: React.FC<SettingsPageProps> = ({
try {
await deleteCallback(accessToken, callbackToDelete);
message.success(`Callback ${callbackToDelete} deleted successfully`);
NotificationsManager.success(`Callback ${callbackToDelete} deleted successfully`);
// Refresh the callbacks list
if (userID && userRole) {
@ -394,7 +418,7 @@ const Settings: React.FC<SettingsPageProps> = ({
setCallbackToDelete(null);
} catch (error) {
console.error("Failed to delete callback:", error);
message.error(`Failed to delete callback: ${error}`);
NotificationsManager.fromBackend(error);
}
};
@ -450,9 +474,14 @@ const Settings: React.FC<SettingsPageProps> = ({
className="text-red-500 hover:text-red-700 cursor-pointer"
/>
<Button
onClick={() =>
serviceHealthCheck(accessToken, callback.name)
}
onClick={async () => {
try {
await serviceHealthCheck(accessToken, callback.name);
NotificationsManager.success("Health check triggered");
} catch (error) {
NotificationsManager.fromBackend(parseErrorMessage(error));
}
}}
className="ml-2"
variant="secondary"
>
@ -551,7 +580,14 @@ const Settings: React.FC<SettingsPageProps> = ({
</Button>
<Button
onClick={() => serviceHealthCheck(accessToken, "slack")}
onClick={async () => {
try {
await serviceHealthCheck(accessToken, "slack");
NotificationsManager.success("Alert test triggered. Test request to slack made - check logs/alerts on slack to verify");
} catch (error) {
NotificationsManager.fromBackend(parseErrorMessage(error));
}
}}
className="mx-2"
>
Test Alerts
@ -579,7 +615,11 @@ const Settings: React.FC<SettingsPageProps> = ({
title="Add Logging Callback"
visible={showAddCallbacksModal}
width={800}
onCancel={() => setShowAddCallbacksModal(false)}
onCancel= {() => {
setShowAddCallbacksModal(false)
setSelectedCallback(null);
setSelectedCallbackParams([]);
}}
footer={null}
>
<a
@ -593,7 +633,7 @@ const Settings: React.FC<SettingsPageProps> = ({
</a>
<Form
form={form}
form={addForm}
onFinish={addNewCallbackCall}
labelCol={{ span: 8 }}
wrapperCol={{ span: 16 }}
@ -659,7 +699,7 @@ const Settings: React.FC<SettingsPageProps> = ({
},
]}
>
<TextInput type="password" />
<Input.Password />
</FormItem>
))}
@ -674,11 +714,14 @@ const Settings: React.FC<SettingsPageProps> = ({
visible={showEditCallback}
width={800}
title={`Edit ${selectedEditCallback?.name} Settings`}
onCancel={() => setShowEditCallback(false)}
onCancel={() => {
setShowEditCallback(false)
setSelectedEditCallback(null);
}}
footer={null}
>
<Form
form={form}
form={editForm}
onFinish={updateCallbackCall}
labelCol={{ span: 8 }}
wrapperCol={{ span: 16 }}
@ -687,13 +730,21 @@ const Settings: React.FC<SettingsPageProps> = ({
<>
{selectedEditCallback &&
selectedEditCallback.variables &&
Object.entries(selectedEditCallback.variables).map(
([param, value]) => (
<FormItem label={param} name={param} key={param}>
<TextInput type="password" defaultValue={value as string} />
</FormItem>
)
)}
Object.entries(selectedEditCallback.variables).map(([param]) => (
<FormItem
label={param}
name={param}
key={param}
rules={[
{
required: true,
message: `Please enter the value for ${param}`,
},
]}
>
<Input.Password />
</FormItem>
))}
</>
<div style={{ textAlign: "right", marginTop: "10px" }}>