mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Merge remote-tracking branch 'origin/litellm_internal_staging' into litellm_standardize_rate_limit_errors-5fb4
This commit is contained in:
commit
a7482095b0
564 changed files with 10293 additions and 3292 deletions
1
.github/workflows/test-unit-proxy-db.yml
vendored
1
.github/workflows/test-unit-proxy-db.yml
vendored
|
|
@ -100,6 +100,7 @@ jobs:
|
|||
test-path: >-
|
||||
tests/proxy_unit_tests/test_auth_checks.py
|
||||
tests/proxy_unit_tests/test_user_api_key_auth.py
|
||||
tests/proxy_unit_tests/test_deprecated_key_grace_period.py
|
||||
workers: 4
|
||||
dist: loadscope
|
||||
timeout: 15
|
||||
|
|
|
|||
|
|
@ -0,0 +1,2 @@
|
|||
-- AlterTable
|
||||
ALTER TABLE "LiteLLM_MCPServerTable" ADD COLUMN IF NOT EXISTS "delegate_auth_to_upstream" BOOLEAN NOT NULL DEFAULT false;
|
||||
|
|
@ -323,6 +323,7 @@ model LiteLLM_MCPServerTable {
|
|||
registration_url String?
|
||||
allow_all_keys Boolean @default(false)
|
||||
available_on_public_internet Boolean @default(true)
|
||||
delegate_auth_to_upstream Boolean @default(false)
|
||||
is_byok Boolean @default(false)
|
||||
byok_description String[] @default([])
|
||||
byok_api_key_help_url String?
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ def _get_redis_kwargs():
|
|||
"retry",
|
||||
}
|
||||
|
||||
include_args = [
|
||||
include_args = {
|
||||
"url",
|
||||
"redis_connect_func",
|
||||
"gcp_service_account",
|
||||
|
|
@ -50,9 +50,9 @@ def _get_redis_kwargs():
|
|||
"azure_client_id",
|
||||
"azure_tenant_id",
|
||||
"azure_client_secret",
|
||||
]
|
||||
}
|
||||
|
||||
available_args = [x for x in arg_spec.args if x not in exclude_args] + include_args
|
||||
available_args = {x for x in arg_spec.args if x not in exclude_args} | include_args
|
||||
|
||||
return available_args
|
||||
|
||||
|
|
@ -84,23 +84,23 @@ def _get_redis_cluster_kwargs(client=None):
|
|||
# Only allow primitive arguments
|
||||
exclude_args = {"self", "connection_pool", "retry", "host", "port", "startup_nodes"}
|
||||
|
||||
available_args = [x for x in arg_spec.args if x not in exclude_args]
|
||||
available_args.append("password")
|
||||
available_args.append("username")
|
||||
available_args.append("ssl")
|
||||
available_args.append("ssl_cert_reqs")
|
||||
available_args.append("ssl_check_hostname")
|
||||
available_args.append("ssl_ca_certs")
|
||||
available_args.append(
|
||||
"redis_connect_func"
|
||||
) # Needed for sync clusters and IAM detection
|
||||
available_args.append("gcp_service_account")
|
||||
available_args.append("gcp_ssl_ca_certs")
|
||||
available_args.append("azure_redis_ad_token")
|
||||
available_args.append("azure_client_id")
|
||||
available_args.append("azure_tenant_id")
|
||||
available_args.append("azure_client_secret")
|
||||
available_args.append("max_connections")
|
||||
available_args = {x for x in arg_spec.args if x not in exclude_args}
|
||||
available_args |= {
|
||||
"password",
|
||||
"username",
|
||||
"ssl",
|
||||
"ssl_cert_reqs",
|
||||
"ssl_check_hostname",
|
||||
"ssl_ca_certs",
|
||||
"redis_connect_func", # Needed for sync clusters and IAM detection
|
||||
"gcp_service_account",
|
||||
"gcp_ssl_ca_certs",
|
||||
"azure_redis_ad_token",
|
||||
"azure_client_id",
|
||||
"azure_tenant_id",
|
||||
"azure_client_secret",
|
||||
"max_connections",
|
||||
}
|
||||
|
||||
return available_args
|
||||
|
||||
|
|
@ -479,10 +479,24 @@ def init_redis_cluster(redis_kwargs) -> redis.RedisCluster:
|
|||
return redis.RedisCluster(startup_nodes=new_startup_nodes, **cluster_kwargs) # type: ignore
|
||||
|
||||
|
||||
def _get_redis_sentinel_connection_kwargs(redis_kwargs: dict) -> dict:
|
||||
connection_kwargs = {}
|
||||
args = _get_redis_kwargs()
|
||||
for arg in redis_kwargs:
|
||||
if arg in args:
|
||||
connection_kwargs[arg] = redis_kwargs[arg]
|
||||
|
||||
return connection_kwargs
|
||||
|
||||
|
||||
def _init_redis_sentinel(redis_kwargs) -> redis.Redis:
|
||||
sentinel_nodes = redis_kwargs.get("sentinel_nodes")
|
||||
sentinel_password = redis_kwargs.get("sentinel_password")
|
||||
service_name = redis_kwargs.get("service_name")
|
||||
connection_kwargs = _get_redis_sentinel_connection_kwargs(redis_kwargs)
|
||||
connection_kwargs.setdefault("socket_timeout", REDIS_SOCKET_TIMEOUT)
|
||||
sentinel_kwargs = dict(connection_kwargs)
|
||||
sentinel_kwargs["password"] = sentinel_password
|
||||
|
||||
if not sentinel_nodes or not service_name:
|
||||
raise ValueError(
|
||||
|
|
@ -494,19 +508,22 @@ def _init_redis_sentinel(redis_kwargs) -> redis.Redis:
|
|||
# Set up the Sentinel client
|
||||
sentinel = redis.Sentinel(
|
||||
sentinel_nodes,
|
||||
socket_timeout=REDIS_SOCKET_TIMEOUT,
|
||||
password=sentinel_password,
|
||||
sentinel_kwargs=sentinel_kwargs,
|
||||
)
|
||||
|
||||
# Return the master instance for the given service
|
||||
|
||||
return sentinel.master_for(service_name)
|
||||
return sentinel.master_for(service_name, **connection_kwargs)
|
||||
|
||||
|
||||
def _init_async_redis_sentinel(redis_kwargs) -> async_redis.Redis:
|
||||
sentinel_nodes = redis_kwargs.get("sentinel_nodes")
|
||||
sentinel_password = redis_kwargs.get("sentinel_password")
|
||||
service_name = redis_kwargs.get("service_name")
|
||||
connection_kwargs = _get_redis_sentinel_connection_kwargs(redis_kwargs)
|
||||
connection_kwargs.setdefault("socket_timeout", REDIS_SOCKET_TIMEOUT)
|
||||
sentinel_kwargs = dict(connection_kwargs)
|
||||
sentinel_kwargs["password"] = sentinel_password
|
||||
|
||||
if not sentinel_nodes or not service_name:
|
||||
raise ValueError(
|
||||
|
|
@ -518,13 +535,12 @@ def _init_async_redis_sentinel(redis_kwargs) -> async_redis.Redis:
|
|||
# Set up the Sentinel client
|
||||
sentinel = async_redis.Sentinel(
|
||||
sentinel_nodes,
|
||||
socket_timeout=REDIS_SOCKET_TIMEOUT,
|
||||
password=sentinel_password,
|
||||
sentinel_kwargs=sentinel_kwargs,
|
||||
)
|
||||
|
||||
# Return the master instance for the given service
|
||||
|
||||
return sentinel.master_for(service_name)
|
||||
return sentinel.master_for(service_name, **connection_kwargs)
|
||||
|
||||
|
||||
def get_redis_client(**env_overrides):
|
||||
|
|
|
|||
|
|
@ -178,6 +178,18 @@ class BudgetManager:
|
|||
return list(self.user_dict.keys())
|
||||
|
||||
def reset_cost(self, user):
|
||||
"""
|
||||
Reset the tracked spend for a user back to zero.
|
||||
|
||||
Clears both the aggregate ``current_cost`` and the per-model
|
||||
``model_cost`` breakdown stored for the given user.
|
||||
|
||||
Args:
|
||||
user: The user identifier whose cost should be reset.
|
||||
|
||||
Returns:
|
||||
dict: ``{"user": <updated user record>}`` reflecting the reset state.
|
||||
"""
|
||||
self.user_dict[user]["current_cost"] = 0
|
||||
self.user_dict[user]["model_cost"] = {}
|
||||
return {"user": self.user_dict[user]}
|
||||
|
|
|
|||
|
|
@ -888,6 +888,15 @@ def log_guardrail_information(func):
|
|||
- pre_call
|
||||
- during_call
|
||||
- post_call
|
||||
|
||||
Some guardrails (e.g. ``block_code_execution``) call
|
||||
``add_standard_logging_guardrail_information_to_request_data`` directly
|
||||
from inside the wrapped function so they can record a richer payload
|
||||
(structured detections, tracing detail) than this decorator's
|
||||
"allow"/"mask"/raw-response default. To avoid double-recording in that
|
||||
case (which would emit two spans, two Datadog records, two spend-log
|
||||
entries, etc.), snapshot the entry count before invocation: if the
|
||||
wrapped function already appended its own entry, skip the auto-record.
|
||||
"""
|
||||
import functools
|
||||
import inspect
|
||||
|
|
@ -907,6 +916,16 @@ def log_guardrail_information(func):
|
|||
return GuardrailEventHooks.post_call
|
||||
return None
|
||||
|
||||
def _count_recorded_guardrail_entries(request_data: dict) -> int:
|
||||
total = 0
|
||||
for container_key in ("metadata", "litellm_metadata"):
|
||||
container = request_data.get(container_key)
|
||||
if isinstance(container, dict):
|
||||
entries = container.get("standard_logging_guardrail_information")
|
||||
if isinstance(entries, list):
|
||||
total += len(entries)
|
||||
return total
|
||||
|
||||
@functools.wraps(func)
|
||||
async def async_wrapper(*args, **kwargs):
|
||||
start_time = datetime.now() # Move start_time inside the wrapper
|
||||
|
|
@ -919,8 +938,11 @@ def log_guardrail_information(func):
|
|||
if func.__name__ == "apply_guardrail" and "inputs" in kwargs:
|
||||
original_inputs = kwargs.get("inputs")
|
||||
|
||||
entries_before = _count_recorded_guardrail_entries(request_data)
|
||||
try:
|
||||
response = await func(*args, **kwargs)
|
||||
if _count_recorded_guardrail_entries(request_data) > entries_before:
|
||||
return response
|
||||
return self._process_response(
|
||||
response=response,
|
||||
request_data=request_data,
|
||||
|
|
@ -931,6 +953,8 @@ def log_guardrail_information(func):
|
|||
original_inputs=original_inputs,
|
||||
)
|
||||
except Exception as e:
|
||||
if _count_recorded_guardrail_entries(request_data) > entries_before:
|
||||
raise
|
||||
return self._process_error(
|
||||
e=e,
|
||||
request_data=request_data,
|
||||
|
|
@ -952,8 +976,11 @@ def log_guardrail_information(func):
|
|||
if func.__name__ == "apply_guardrail" and "inputs" in kwargs:
|
||||
original_inputs = kwargs.get("inputs")
|
||||
|
||||
entries_before = _count_recorded_guardrail_entries(request_data)
|
||||
try:
|
||||
response = func(*args, **kwargs)
|
||||
if _count_recorded_guardrail_entries(request_data) > entries_before:
|
||||
return response
|
||||
return self._process_response(
|
||||
response=response,
|
||||
request_data=request_data,
|
||||
|
|
@ -962,6 +989,8 @@ def log_guardrail_information(func):
|
|||
original_inputs=original_inputs,
|
||||
)
|
||||
except Exception as e:
|
||||
if _count_recorded_guardrail_entries(request_data) > entries_before:
|
||||
raise
|
||||
return self._process_error(
|
||||
e=e,
|
||||
request_data=request_data,
|
||||
|
|
|
|||
|
|
@ -237,7 +237,14 @@ class OpenTelemetry(CustomLogger):
|
|||
not isinstance(cb, OpenTelemetry) for cb in litellm.service_callback
|
||||
):
|
||||
litellm.service_callback.append(self)
|
||||
setattr(proxy_server, "open_telemetry_logger", self)
|
||||
# avoid proxy logger ownership being overwritten by later
|
||||
# handlers. Multiple integrations (default OTEL, Langfuse OTEL,
|
||||
# Arize OTEL, etc.) may initialize in sequence; without this guard,
|
||||
# the last one silently replaces the first and breaks expected
|
||||
# routing for proxy_server.open_telemetry_logger consumers.
|
||||
# Behavior: first-registered wins.
|
||||
if getattr(proxy_server, "open_telemetry_logger", None) is None:
|
||||
setattr(proxy_server, "open_telemetry_logger", self)
|
||||
|
||||
def _get_or_create_provider(
|
||||
self,
|
||||
|
|
@ -794,12 +801,100 @@ class OpenTelemetry(CustomLogger):
|
|||
# End of Team/Key Based Logging Control Flow
|
||||
#########################################################
|
||||
|
||||
def _emit_once(self, kwargs: dict, *scope: object) -> bool:
|
||||
"""Return True the first time this handler is asked to emit a span
|
||||
for the given (handler, scope) on this kwargs; False on repeats.
|
||||
|
||||
Used to suppress duplicate span emission for two distinct patterns:
|
||||
|
||||
1. **Handler-level dual-fire**: streaming code paths trigger both
|
||||
the sync and async callback for one request, so ``_handle_success``
|
||||
/ ``_handle_failure`` would otherwise produce two
|
||||
``litellm_request`` spans. Scope: ``("success",)`` / ``("failure",)``.
|
||||
2. **Payload-driven multi-entrypoint emission**: a span loop that
|
||||
reads entries from ``standard_logging_payload`` (currently only
|
||||
guardrails) is invoked from multiple lifecycle points
|
||||
(post-call hooks, success callback, failure callback). The list
|
||||
can be re-read with mutated entries between calls, so dedupe
|
||||
must be at entry granularity. Scope: the entry's stable identity.
|
||||
|
||||
``scope`` parts can be any hashable identity. The marker is stored
|
||||
in ``kwargs["litellm_params"]["metadata"]["_otel_internal"]`` so it
|
||||
is request-local (kwargs is shared across the sync/async callbacks
|
||||
and lifecycle hooks for one request).
|
||||
"""
|
||||
litellm_params = kwargs.get("litellm_params")
|
||||
if not isinstance(litellm_params, dict):
|
||||
litellm_params = {}
|
||||
kwargs["litellm_params"] = litellm_params
|
||||
|
||||
_metadata = litellm_params.get("metadata")
|
||||
if not isinstance(_metadata, dict):
|
||||
_metadata = {}
|
||||
litellm_params["metadata"] = _metadata
|
||||
|
||||
_otel_internal = _metadata.get("_otel_internal")
|
||||
if not isinstance(_otel_internal, dict):
|
||||
_otel_internal = {}
|
||||
_metadata["_otel_internal"] = _otel_internal
|
||||
|
||||
spans_logged = _otel_internal.get("spans_logged")
|
||||
if not isinstance(spans_logged, dict):
|
||||
spans_logged = {}
|
||||
_otel_internal["spans_logged"] = spans_logged
|
||||
|
||||
dedupe_key = (self.__class__.__name__, id(self), *scope)
|
||||
if spans_logged.get(dedupe_key) is True:
|
||||
return False
|
||||
|
||||
spans_logged[dedupe_key] = True
|
||||
return True
|
||||
|
||||
def _end_proxy_span_from_kwargs(self, kwargs: dict, end_time) -> None:
|
||||
"""Close the proxy-level parent span if it is still recording.
|
||||
|
||||
This helper retrieves the proxy span directly from kwargs metadata
|
||||
and closes it after all child spans have been recorded.
|
||||
|
||||
Only called from the success path. The failure path deliberately
|
||||
leaves the proxy span open so ``async_post_call_failure_hook`` can
|
||||
append the ``"Failed Proxy Server Request"`` child span before
|
||||
closing it.
|
||||
|
||||
Only spans named ``LITELLM_PROXY_REQUEST_SPAN_NAME`` are closed —
|
||||
externally provided spans must not be closed by LiteLLM.
|
||||
"""
|
||||
litellm_params = kwargs.get("litellm_params", {}) or {}
|
||||
_metadata = litellm_params.get("metadata", {}) or {}
|
||||
proxy_span = _metadata.get("litellm_parent_otel_span", None)
|
||||
if (
|
||||
proxy_span is not None
|
||||
and getattr(proxy_span, "name", None) == LITELLM_PROXY_REQUEST_SPAN_NAME
|
||||
and hasattr(proxy_span, "is_recording")
|
||||
and proxy_span.is_recording()
|
||||
):
|
||||
proxy_span.end(end_time=self._to_ns(end_time))
|
||||
|
||||
def _handle_success(self, kwargs, response_obj, start_time, end_time):
|
||||
"""Create the litellm_request span then close the proxy span."""
|
||||
verbose_logger.debug(
|
||||
"OpenTelemetry Logger: Logging kwargs: %s, OTEL config settings=%s",
|
||||
kwargs,
|
||||
self.config,
|
||||
)
|
||||
|
||||
# sync + async success handlers can both fire for one
|
||||
# request (notably in streaming code paths). Guard against duplicate
|
||||
# span writes — but still close the proxy span on the skip path so
|
||||
# the trace doesn't leak an open root span.
|
||||
if not self._emit_once(kwargs, "success"):
|
||||
verbose_logger.debug(
|
||||
"OpenTelemetry: skipping duplicate success span for handler=%s",
|
||||
self.__class__.__name__,
|
||||
)
|
||||
self._end_proxy_span_from_kwargs(kwargs, end_time)
|
||||
return
|
||||
|
||||
ctx, parent_span = self._get_span_context(kwargs)
|
||||
|
||||
if self.config.ignore_context_propagation:
|
||||
|
|
@ -859,7 +954,7 @@ class OpenTelemetry(CustomLogger):
|
|||
|
||||
# 6. Do NOT end parent span - it should be managed by its creator
|
||||
# External spans (from Langfuse, user code, HTTP headers, global context) must not be closed by LiteLLM
|
||||
# However, proxy-created spans should be closed here
|
||||
# However, proxy-created spans should be closed here.
|
||||
if (
|
||||
parent_span is not None
|
||||
and hasattr(parent_span, "name")
|
||||
|
|
@ -867,6 +962,11 @@ class OpenTelemetry(CustomLogger):
|
|||
):
|
||||
parent_span.end(end_time=self._to_ns(end_time))
|
||||
|
||||
# close the proxy span explicitly from kwargs metadata
|
||||
# after all child spans (litellm_request, guardrail, raw_request)
|
||||
# have been fully recorded and exported.
|
||||
self._end_proxy_span_from_kwargs(kwargs, end_time)
|
||||
|
||||
def _start_primary_span(
|
||||
self,
|
||||
kwargs,
|
||||
|
|
@ -1296,6 +1396,21 @@ class OpenTelemetry(CustomLogger):
|
|||
for guardrail_information in guardrail_information_list:
|
||||
start_time_float = guardrail_information.get("start_time")
|
||||
end_time_float = guardrail_information.get("end_time")
|
||||
|
||||
# ``_create_guardrail_span`` is called from three lifecycle
|
||||
# points (``async_post_call_success_hook``, ``_handle_success``,
|
||||
# ``_handle_failure``) and re-reads the (mutating) entry list
|
||||
# each time. Dedupe at entry granularity so a single real
|
||||
# guardrail invocation produces exactly one span per handler.
|
||||
if not self._emit_once(
|
||||
kwargs,
|
||||
"guardrail",
|
||||
guardrail_information.get("guardrail_name"),
|
||||
start_time_float,
|
||||
guardrail_information.get("guardrail_mode"),
|
||||
):
|
||||
continue
|
||||
|
||||
start_time_datetime = datetime.now()
|
||||
if start_time_float is not None:
|
||||
start_time_datetime = datetime.fromtimestamp(start_time_float)
|
||||
|
|
@ -1349,6 +1464,21 @@ class OpenTelemetry(CustomLogger):
|
|||
kwargs,
|
||||
self.config,
|
||||
)
|
||||
|
||||
# sync + async failure handlers can both fire for one
|
||||
# request (notably in streaming code paths), producing two
|
||||
# semantically identical ERROR spans. Unlike the success path, the
|
||||
# proxy span is intentionally left open here so that
|
||||
# ``async_post_call_failure_hook`` can append the
|
||||
# "Failed Proxy Server Request" child span before closing it —
|
||||
# there is no proxy-span side-effect to preserve on the skip path.
|
||||
if not self._emit_once(kwargs, "failure"):
|
||||
verbose_logger.debug(
|
||||
"OpenTelemetry: skipping duplicate failure span for handler=%s",
|
||||
self.__class__.__name__,
|
||||
)
|
||||
return
|
||||
|
||||
_parent_context, parent_otel_span = self._get_span_context(kwargs)
|
||||
|
||||
if self.config.ignore_context_propagation:
|
||||
|
|
@ -2188,7 +2318,7 @@ class OpenTelemetry(CustomLogger):
|
|||
verbose_logger.debug(
|
||||
"OpenTelemetry: Using explicit parent span from metadata"
|
||||
)
|
||||
return trace.set_span_in_context(parent_otel_span), parent_otel_span
|
||||
return trace.set_span_in_context(parent_otel_span), None
|
||||
|
||||
# Priority 2: HTTP traceparent header
|
||||
if traceparent is not None:
|
||||
|
|
|
|||
|
|
@ -1226,6 +1226,17 @@ class PrometheusLogger(CustomLogger):
|
|||
label_context=label_context,
|
||||
)
|
||||
|
||||
# Provider-agnostic fallback: providers like Bedrock and Vertex don't return
|
||||
# x-ratelimit-remaining-* headers, so the gauges above only fire for OpenAI /
|
||||
# Anthropic / Azure. When the proxy router has tpm/rpm configured for the
|
||||
# model_group, derive remaining from configured-limit minus current usage so
|
||||
# the same metric is populated for any provider.
|
||||
await self._async_set_router_remaining_metrics(
|
||||
standard_logging_payload=standard_logging_payload, # type: ignore
|
||||
enum_values=enum_values,
|
||||
label_context=label_context,
|
||||
)
|
||||
|
||||
# cache metrics
|
||||
self._increment_cache_metrics(
|
||||
standard_logging_payload=standard_logging_payload, # type: ignore
|
||||
|
|
@ -2198,6 +2209,99 @@ class PrometheusLogger(CustomLogger):
|
|||
)
|
||||
self.litellm_deployment_rpm_limit.labels(**_labels).set(rpm)
|
||||
|
||||
async def _async_set_router_remaining_metrics(
|
||||
self,
|
||||
standard_logging_payload: StandardLoggingPayload,
|
||||
enum_values: UserAPIKeyLabelValues,
|
||||
label_context: Optional[PrometheusLabelFactoryContext] = None,
|
||||
) -> None:
|
||||
"""
|
||||
Populate ``litellm_remaining_tokens_metric`` /
|
||||
``litellm_remaining_requests_metric`` from the router's internal usage
|
||||
counters when the upstream provider did not return
|
||||
``x-ratelimit-remaining-*`` response headers.
|
||||
|
||||
OpenAI / Anthropic / Azure return remaining tokens/requests in response
|
||||
headers, but Bedrock and Vertex AI do not. This fallback computes
|
||||
``configured_limit - current_usage`` via
|
||||
``Router.get_remaining_model_group_usage`` so the same gauges are
|
||||
emitted for every provider when tpm/rpm is configured on the
|
||||
deployment.
|
||||
"""
|
||||
try:
|
||||
additional_headers = (
|
||||
standard_logging_payload.get("hidden_params", {}) or {}
|
||||
).get("additional_headers") or {}
|
||||
|
||||
already_have_tokens = (
|
||||
additional_headers.get("x_ratelimit_remaining_tokens") is not None
|
||||
)
|
||||
already_have_requests = (
|
||||
additional_headers.get("x_ratelimit_remaining_requests") is not None
|
||||
)
|
||||
if already_have_tokens and already_have_requests:
|
||||
return
|
||||
|
||||
model_group = standard_logging_payload.get("model_group")
|
||||
if not model_group:
|
||||
return
|
||||
|
||||
try:
|
||||
from litellm.proxy.proxy_server import llm_router
|
||||
except ImportError:
|
||||
llm_router = None
|
||||
|
||||
if llm_router is None:
|
||||
return
|
||||
|
||||
try:
|
||||
remaining_usage = await llm_router.get_remaining_model_group_usage(
|
||||
model_group
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(
|
||||
"Prometheus: get_remaining_model_group_usage failed for "
|
||||
"model_group=%s: %s",
|
||||
model_group,
|
||||
e,
|
||||
)
|
||||
return
|
||||
|
||||
if not remaining_usage:
|
||||
return
|
||||
|
||||
remaining_tokens = remaining_usage.get("x-ratelimit-remaining-tokens")
|
||||
remaining_requests = remaining_usage.get("x-ratelimit-remaining-requests")
|
||||
|
||||
if not already_have_tokens and remaining_tokens is not None:
|
||||
_labels = prometheus_label_factory(
|
||||
supported_enum_labels=self.get_labels_for_metric(
|
||||
metric_name="litellm_remaining_tokens_metric"
|
||||
),
|
||||
enum_values=enum_values,
|
||||
label_context=label_context,
|
||||
)
|
||||
self.litellm_remaining_tokens_metric.labels(**_labels).set(
|
||||
remaining_tokens
|
||||
)
|
||||
|
||||
if not already_have_requests and remaining_requests is not None:
|
||||
_labels = prometheus_label_factory(
|
||||
supported_enum_labels=self.get_labels_for_metric(
|
||||
metric_name="litellm_remaining_requests_metric"
|
||||
),
|
||||
enum_values=enum_values,
|
||||
label_context=label_context,
|
||||
)
|
||||
self.litellm_remaining_requests_metric.labels(**_labels).set(
|
||||
remaining_requests
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.exception(
|
||||
"Prometheus Error: _async_set_router_remaining_metrics. "
|
||||
"Exception occured - {}".format(str(e))
|
||||
)
|
||||
|
||||
def set_llm_deployment_success_metrics(
|
||||
self,
|
||||
request_kwargs: dict,
|
||||
|
|
|
|||
|
|
@ -53,8 +53,19 @@ def process_audio_file(audio_file: FileTypes) -> ProcessedAudioFile:
|
|||
# Raw bytes
|
||||
filename = "audio.wav"
|
||||
file_content = bytes(audio_file)
|
||||
elif isinstance(audio_file, (str, os.PathLike)):
|
||||
# File path or PathLike
|
||||
elif isinstance(audio_file, str):
|
||||
# Bare strings are rejected — see extract_file_data for the same
|
||||
# rationale: in a proxy request handler the string is
|
||||
# attacker-controlled, and opening it as a path is an arbitrary
|
||||
# file read.
|
||||
raise ValueError(
|
||||
"process_audio_file does not accept bare str inputs. Pass bytes, "
|
||||
"an open file handle, a (filename, content) tuple, or a "
|
||||
"pathlib.Path."
|
||||
)
|
||||
elif isinstance(audio_file, os.PathLike):
|
||||
# File path or PathLike — PathLike is a Python-level type that
|
||||
# HTTP form values can't fabricate.
|
||||
file_path = str(audio_file)
|
||||
with open(file_path, "rb") as f:
|
||||
file_content = f.read()
|
||||
|
|
@ -66,8 +77,14 @@ def process_audio_file(audio_file: FileTypes) -> ProcessedAudioFile:
|
|||
content = audio_file[1]
|
||||
if isinstance(content, (bytes, bytearray)):
|
||||
file_content = bytes(content)
|
||||
elif isinstance(content, (str, os.PathLike)):
|
||||
# File path or PathLike
|
||||
elif isinstance(content, str):
|
||||
raise ValueError(
|
||||
"process_audio_file does not accept bare str tuple "
|
||||
"contents. Pass bytes, an open file handle, or a "
|
||||
"pathlib.Path."
|
||||
)
|
||||
elif isinstance(content, os.PathLike):
|
||||
# PathLike: SDK convenience for local-file uploads.
|
||||
with open(str(content), "rb") as f:
|
||||
file_content = f.read()
|
||||
elif hasattr(content, "read"):
|
||||
|
|
@ -149,7 +166,14 @@ def get_audio_file_content_hash(file_obj: FileTypes) -> str:
|
|||
try:
|
||||
if isinstance(file_content_obj, (bytes, bytearray)):
|
||||
file_content = bytes(file_content_obj)
|
||||
elif isinstance(file_content_obj, (str, os.PathLike)):
|
||||
elif isinstance(file_content_obj, str):
|
||||
# Bare strings are not treated as file paths in this helper —
|
||||
# the cache-key path is reached from request handlers where the
|
||||
# value is attacker-controlled. Fall back to hashing the string
|
||||
# itself rather than opening it.
|
||||
fallback_filename = file_content_obj
|
||||
file_content = None
|
||||
elif isinstance(file_content_obj, os.PathLike):
|
||||
try:
|
||||
with open(str(file_content_obj), "rb") as f:
|
||||
file_content = f.read()
|
||||
|
|
@ -229,8 +253,15 @@ def calculate_request_duration(file: FileTypes) -> Optional[float]:
|
|||
if isinstance(file, (bytes, bytearray)):
|
||||
# Raw bytes
|
||||
file_content = bytes(file)
|
||||
elif isinstance(file, (str, os.PathLike)):
|
||||
# File path
|
||||
elif isinstance(file, str):
|
||||
# Bare strings are rejected — see extract_file_data.
|
||||
raise ValueError(
|
||||
"calculate_request_duration does not accept bare str inputs. "
|
||||
"Pass bytes, an open file handle, a (filename, content) "
|
||||
"tuple, or a pathlib.Path."
|
||||
)
|
||||
elif isinstance(file, os.PathLike):
|
||||
# File path (PathLike): SDK convenience.
|
||||
with open(str(file), "rb") as f:
|
||||
file_content = f.read()
|
||||
elif isinstance(file, tuple):
|
||||
|
|
|
|||
|
|
@ -755,14 +755,25 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData:
|
|||
else:
|
||||
file_content = file_data
|
||||
# Convert content to bytes
|
||||
if isinstance(file_content, (str, PathLike)):
|
||||
# If it's a path, open and read the file
|
||||
# Extract filename from path if not already set
|
||||
if isinstance(file_content, str):
|
||||
# Bare string inputs are rejected: when this helper runs in a proxy
|
||||
# request handler the string came from an attacker-controlled form
|
||||
# field, and opening it as a path is an arbitrary file read on the
|
||||
# proxy host. SDK callers who want to upload from a path should
|
||||
# either pass a pathlib.Path (a PathLike instance — see the branch
|
||||
# below) or open the file themselves and pass the handle / bytes.
|
||||
raise ValueError(
|
||||
"extract_file_data does not accept bare str inputs. Pass bytes, "
|
||||
"an open file handle, a (filename, content) tuple, or a "
|
||||
"pathlib.Path. To upload a local file from a path, call "
|
||||
"open(path, 'rb') yourself."
|
||||
)
|
||||
if isinstance(file_content, PathLike):
|
||||
# PathLike (pathlib.Path) is a Python-level type that HTTP form
|
||||
# values can't fabricate. Treat as a local file path for SDK
|
||||
# convenience.
|
||||
if filename is None:
|
||||
if isinstance(file_content, PathLike):
|
||||
filename = Path(file_content).name
|
||||
else:
|
||||
filename = Path(str(file_content)).name
|
||||
filename = Path(file_content).name
|
||||
with open(file_content, "rb") as f:
|
||||
content = f.read()
|
||||
elif isinstance(file_content, io.IOBase):
|
||||
|
|
|
|||
|
|
@ -4977,8 +4977,9 @@ class BedrockConverseMessagesProcessor:
|
|||
)
|
||||
if reasoning_text and not reasoning_text.get("signature"):
|
||||
reasoning_text_text = reasoning_text["text"]
|
||||
assistants_part = BedrockContentBlock(text=reasoning_text_text)
|
||||
assistant_parts.append(assistants_part)
|
||||
if reasoning_text_text.strip():
|
||||
assistants_part = BedrockContentBlock(text=reasoning_text_text)
|
||||
assistant_parts.append(assistants_part)
|
||||
else:
|
||||
filtered_thinking_blocks.append(block)
|
||||
if len(filtered_thinking_blocks) > 0:
|
||||
|
|
|
|||
|
|
@ -1299,9 +1299,18 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
else truncated_name
|
||||
)
|
||||
|
||||
# Strip Gemini thought-signature suffix from id (mirrors streaming
|
||||
# path below); base64 chars (+ / =) violate Anthropic's
|
||||
# `^[a-zA-Z0-9_-]+$` tool_use.id pattern when replayed.
|
||||
raw_id = tool_call.id or ""
|
||||
base_id = (
|
||||
raw_id.split(THOUGHT_SIGNATURE_SEPARATOR, 1)[0]
|
||||
if THOUGHT_SIGNATURE_SEPARATOR in raw_id
|
||||
else raw_id
|
||||
)
|
||||
tool_use_block = AnthropicResponseContentBlockToolUse(
|
||||
type="tool_use",
|
||||
id=tool_call.id,
|
||||
id=base_id,
|
||||
name=original_name,
|
||||
input=parse_tool_call_arguments(
|
||||
tool_call.function.arguments,
|
||||
|
|
|
|||
|
|
@ -3,8 +3,10 @@ from typing import Optional, cast
|
|||
import httpx
|
||||
|
||||
import litellm
|
||||
from litellm.llms.azure.common_utils import BaseAzureLLM
|
||||
from litellm.llms.openai.image_edit.transformation import OpenAIImageEditConfig
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.utils import _add_path_to_api_base
|
||||
|
||||
|
||||
|
|
@ -30,20 +32,42 @@ class AzureImageEditConfig(OpenAIImageEditConfig):
|
|||
litellm_params: Optional[dict] = None,
|
||||
api_base: Optional[str] = None,
|
||||
) -> dict:
|
||||
api_key = (
|
||||
api_key
|
||||
or litellm.api_key
|
||||
or litellm.azure_key
|
||||
or get_secret_str("AZURE_OPENAI_API_KEY")
|
||||
or get_secret_str("AZURE_API_KEY")
|
||||
)
|
||||
"""
|
||||
Validate Azure environment and set up authentication headers.
|
||||
|
||||
headers.update(
|
||||
{
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
}
|
||||
Delegates to ``BaseAzureLLM._base_validate_azure_environment`` so the
|
||||
Azure image-edit route uses the same auth resolution as every other
|
||||
Azure provider (videos, vector_stores, responses, containers, ...):
|
||||
|
||||
- prefers the Azure-style ``api-key`` header when an API key is available
|
||||
- falls back to ``Authorization: Bearer <azure_ad_token>`` only when AAD
|
||||
auth is configured
|
||||
|
||||
The previous implementation unconditionally set
|
||||
``Authorization: Bearer <api_key>``, which is correct for OpenAI direct
|
||||
but not for Azure OpenAI / API Management gateways that expect the
|
||||
``api-key`` header. Subscription-key-based deployments (e.g., behind
|
||||
Azure APIM) responded with ``401 "Access denied due to missing
|
||||
subscription key"``.
|
||||
|
||||
API-key precedence (matches ``AzureVideosConfig``):
|
||||
|
||||
- ``litellm_params["api_key"]`` is the source of truth.
|
||||
- The positional ``api_key`` kwarg only fills in when
|
||||
``litellm_params["api_key"]`` is empty.
|
||||
- This is a deliberate change from the old ``or`` chain (where the
|
||||
positional ``api_key`` argument won) so behavior matches every other
|
||||
Azure ``validate_environment`` implementation. In production the only
|
||||
caller (``llm_http_handler.image_edit``) sources both values from
|
||||
the same ``litellm_params.api_key``, so the precedence only matters
|
||||
for direct callers of this method.
|
||||
"""
|
||||
params = GenericLiteLLMParams(**(litellm_params or {}))
|
||||
if api_key is not None and params.api_key is None:
|
||||
params.api_key = api_key
|
||||
return BaseAzureLLM._base_validate_azure_environment(
|
||||
headers=headers, litellm_params=params
|
||||
)
|
||||
return headers
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -485,11 +485,16 @@ class MaskedHTTPStatusError(httpx.HTTPStatusError):
|
|||
if k.lower() not in ("content-encoding", "content-length")
|
||||
}
|
||||
|
||||
try:
|
||||
request_content = original_error.request.content
|
||||
except httpx.RequestNotRead:
|
||||
request_content = b""
|
||||
|
||||
masked_request = httpx.Request(
|
||||
method=original_error.request.method,
|
||||
url=masked_url,
|
||||
headers=original_error.request.headers,
|
||||
content=original_error.request.content,
|
||||
content=request_content,
|
||||
)
|
||||
|
||||
super().__init__(
|
||||
|
|
|
|||
|
|
@ -241,10 +241,13 @@ class FireworksAIConfig(OpenAIGPTConfig):
|
|||
disable_add_transform_inline_image_block=disable_add_transform_inline_image_block,
|
||||
)
|
||||
filter_value_from_dict(cast(dict, message), "cache_control")
|
||||
# Remove fields not permitted by FireworksAI that may cause:
|
||||
# "Not permitted, field: 'messages[n].provider_specific_fields'"
|
||||
if isinstance(message, dict) and "provider_specific_fields" in message:
|
||||
cast(dict, message).pop("provider_specific_fields", None)
|
||||
# Remove fields not permitted by FireworksAI (additionalProperties: false
|
||||
# on their ChatMessage schema) that may cause:
|
||||
# "Extra inputs are not permitted, field: 'messages[n].<field>'"
|
||||
if isinstance(message, dict):
|
||||
m = cast(dict, message)
|
||||
m.pop("provider_specific_fields", None)
|
||||
m.pop("thinking_blocks", None)
|
||||
|
||||
return messages
|
||||
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@
|
|||
Transformation for Calling Google models in their native format.
|
||||
"""
|
||||
|
||||
from copy import deepcopy
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional, Tuple, Union, cast
|
||||
|
||||
import httpx
|
||||
|
|
@ -11,6 +12,10 @@ from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging
|
|||
from litellm.llms.base_llm.google_genai.transformation import (
|
||||
BaseGoogleGenAIGenerateContentConfig,
|
||||
)
|
||||
from litellm.llms.vertex_ai.common_utils import (
|
||||
_build_vertex_schema,
|
||||
supports_response_json_schema,
|
||||
)
|
||||
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import VertexLLM
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
|
|
@ -302,6 +307,52 @@ class GoogleGenAIConfig(BaseGoogleGenAIGenerateContentConfig, VertexLLM):
|
|||
litellm_params=litellm_params,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_response_schema(
|
||||
generate_content_config_dict: Dict, model: str
|
||||
) -> None:
|
||||
schema_key = next(
|
||||
(
|
||||
k
|
||||
for k in ("responseSchema", "response_schema")
|
||||
if k in generate_content_config_dict
|
||||
),
|
||||
None,
|
||||
)
|
||||
json_schema_key = next(
|
||||
(
|
||||
k
|
||||
for k in ("responseJsonSchema", "response_json_schema")
|
||||
if k in generate_content_config_dict
|
||||
),
|
||||
None,
|
||||
)
|
||||
|
||||
if schema_key is None:
|
||||
return
|
||||
|
||||
value = generate_content_config_dict[schema_key]
|
||||
if not isinstance(value, dict):
|
||||
return
|
||||
|
||||
if supports_response_json_schema(model):
|
||||
if json_schema_key is not None:
|
||||
generate_content_config_dict.pop(schema_key)
|
||||
return
|
||||
generate_content_config_dict.pop(schema_key)
|
||||
new_json_schema_key = (
|
||||
"response_json_schema"
|
||||
if schema_key == "response_schema"
|
||||
else "responseJsonSchema"
|
||||
)
|
||||
generate_content_config_dict[new_json_schema_key] = value
|
||||
else:
|
||||
if json_schema_key is not None:
|
||||
generate_content_config_dict.pop(json_schema_key)
|
||||
generate_content_config_dict[schema_key] = _build_vertex_schema(
|
||||
parameters=deepcopy(value), add_property_ordering=True
|
||||
)
|
||||
|
||||
def transform_generate_content_request(
|
||||
self,
|
||||
model: str,
|
||||
|
|
@ -315,6 +366,8 @@ class GoogleGenAIConfig(BaseGoogleGenAIGenerateContentConfig, VertexLLM):
|
|||
GenerateContentRequestDict,
|
||||
)
|
||||
|
||||
self._normalize_response_schema(generate_content_config_dict, model)
|
||||
|
||||
typed_generate_content_request = GenerateContentRequestDict(
|
||||
model=model,
|
||||
contents=contents,
|
||||
|
|
|
|||
|
|
@ -507,10 +507,10 @@ class OllamaChatCompletionResponseIterator(BaseModelResponseIterator):
|
|||
# PROCESS REASONING CONTENT
|
||||
reasoning_content: Optional[str] = None
|
||||
content: Optional[str] = None
|
||||
if chunk["message"].get("thinking") is not None:
|
||||
if chunk["message"].get("thinking"):
|
||||
reasoning_content = chunk["message"].get("thinking")
|
||||
self.started_reasoning_content = True
|
||||
elif chunk["message"].get("content") is not None:
|
||||
if chunk["message"].get("content"):
|
||||
if (
|
||||
self.started_reasoning_content
|
||||
and not self.finished_reasoning_content
|
||||
|
|
|
|||
|
|
@ -108,7 +108,7 @@ class OllamaModelInfo(BaseLLMModelInfo):
|
|||
continue
|
||||
nm = entry.get("name") or entry.get("model")
|
||||
if isinstance(nm, str):
|
||||
names.add(nm)
|
||||
names.add(nm if nm.startswith("ollama/") else f"ollama/{nm}")
|
||||
except Exception as e:
|
||||
verbose_logger.warning(f"Error retrieving ollama tag endpoint: {e}")
|
||||
# If tags endpoint fails, fall back to static list
|
||||
|
|
|
|||
|
|
@ -79,6 +79,9 @@ class VertexAIGoogleGenAIConfig(GoogleGenAIConfig):
|
|||
Transform the generate content request for Vertex AI.
|
||||
Since Vertex AI natively supports Google GenAI format, we can pass most fields directly.
|
||||
"""
|
||||
if generate_content_config_dict:
|
||||
self._normalize_response_schema(generate_content_config_dict, model)
|
||||
|
||||
# Build the request in Google GenAI format that Vertex AI expects
|
||||
result = {
|
||||
"model": model,
|
||||
|
|
|
|||
|
|
@ -1528,7 +1528,11 @@ def completion( # type: ignore # noqa: PLR0915
|
|||
"logit_bias": logit_bias,
|
||||
"user": user,
|
||||
# params to identify the model
|
||||
"model": model,
|
||||
"model": (
|
||||
model_info.get("base_model")
|
||||
if isinstance(model_info, dict) and model_info.get("base_model")
|
||||
else model
|
||||
),
|
||||
"custom_llm_provider": custom_llm_provider,
|
||||
"response_format": response_format,
|
||||
"seed": seed,
|
||||
|
|
|
|||
|
|
@ -3521,7 +3521,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"azure/gpt-4o-mini-transcribe": {
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_audio_token": 1.25e-06,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
@ -3596,7 +3596,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"azure/gpt-4o-transcribe": {
|
||||
"input_cost_per_audio_token": 6e-06,
|
||||
"input_cost_per_audio_token": 2.5e-06,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
@ -3608,7 +3608,7 @@
|
|||
]
|
||||
},
|
||||
"azure/gpt-4o-transcribe-diarize": {
|
||||
"input_cost_per_audio_token": 6e-06,
|
||||
"input_cost_per_audio_token": 2.5e-06,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
@ -8974,7 +8974,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"gpt-4o-transcribe-diarize": {
|
||||
"input_cost_per_audio_token": 6e-06,
|
||||
"input_cost_per_audio_token": 2.5e-06,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
@ -15551,14 +15551,17 @@
|
|||
"uses_embed_content": true
|
||||
},
|
||||
"vertex_ai/gemini-embedding-2-preview": {
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"input_cost_per_audio_per_second": 0.00016,
|
||||
"input_cost_per_image": 0.00012,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_video_per_second": 0.00079,
|
||||
"litellm_provider": "vertex_ai",
|
||||
"max_input_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "embedding",
|
||||
"output_cost_per_token": 0,
|
||||
"output_vector_size": 3072,
|
||||
"source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal",
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
|
||||
"supports_multimodal": true,
|
||||
"uses_embed_content": true
|
||||
},
|
||||
|
|
@ -15573,7 +15576,7 @@
|
|||
"mode": "embedding",
|
||||
"output_cost_per_token": 0,
|
||||
"output_vector_size": 3072,
|
||||
"source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal",
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing",
|
||||
"supports_multimodal": true,
|
||||
"uses_embed_content": true
|
||||
},
|
||||
|
|
@ -18988,7 +18991,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"gpt-4o-mini-transcribe": {
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_audio_token": 1.25e-06,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
@ -19118,7 +19121,7 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"gpt-4o-transcribe": {
|
||||
"input_cost_per_audio_token": 6e-06,
|
||||
"input_cost_per_audio_token": 2.5e-06,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
@ -21104,6 +21107,38 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-realtime-2": {
|
||||
"cache_creation_input_audio_token_cost": 4e-07,
|
||||
"cache_read_input_token_cost": 4e-07,
|
||||
"input_cost_per_audio_token": 3.2e-05,
|
||||
"input_cost_per_image": 5e-06,
|
||||
"input_cost_per_token": 4e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 32000,
|
||||
"max_output_tokens": 4096,
|
||||
"max_tokens": 4096,
|
||||
"mode": "chat",
|
||||
"output_cost_per_audio_token": 6.4e-05,
|
||||
"output_cost_per_token": 1.6e-05,
|
||||
"supported_endpoints": [
|
||||
"/v1/realtime"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text",
|
||||
"audio"
|
||||
],
|
||||
"supports_audio_input": true,
|
||||
"supports_audio_output": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"gpt-realtime-mini": {
|
||||
"cache_creation_input_audio_token_cost": 3e-07,
|
||||
"cache_read_input_audio_token_cost": 3e-07,
|
||||
|
|
@ -38898,7 +38933,7 @@
|
|||
]
|
||||
},
|
||||
"gpt-4o-mini-transcribe-2025-03-20": {
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_audio_token": 1.25e-06,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
@ -38910,7 +38945,7 @@
|
|||
]
|
||||
},
|
||||
"gpt-4o-mini-transcribe-2025-12-15": {
|
||||
"input_cost_per_audio_token": 3e-06,
|
||||
"input_cost_per_audio_token": 1.25e-06,
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 16000,
|
||||
|
|
|
|||
|
|
@ -10,7 +10,6 @@ import os
|
|||
import re
|
||||
from functools import partial
|
||||
from io import IOBase
|
||||
from pathlib import Path
|
||||
from typing import Any, Coroutine, Dict, Optional, Union
|
||||
|
||||
import httpx
|
||||
|
|
@ -376,11 +375,13 @@ def convert_file_document_to_url_document(document: Dict[str, Any]) -> Dict[str,
|
|||
with an inline base64 data URI.
|
||||
|
||||
Accepts document dicts like:
|
||||
{"type": "file", "file": "/path/to/document.pdf"} # file path string
|
||||
{"type": "file", "file": Path("/path/to/doc.pdf")} # pathlib.Path
|
||||
{"type": "file", "file": <binary file-like object>} # file-like object (BinaryIO)
|
||||
{"type": "file", "file": b"raw bytes"} # raw bytes
|
||||
|
||||
Bare ``str`` paths are not accepted — pass a ``pathlib.Path`` or
|
||||
``open(path, "rb")`` instead. See the str check below for the rationale.
|
||||
|
||||
Returns:
|
||||
{"type": "document_url", "document_url": "data:<mime>;base64,<data>"}
|
||||
or {"type": "image_url", "image_url": "data:<mime>;base64,<data>"}
|
||||
|
|
@ -389,14 +390,28 @@ def convert_file_document_to_url_document(document: Dict[str, Any]) -> Dict[str,
|
|||
if file_input is None:
|
||||
raise ValueError(
|
||||
"document with type='file' must include a 'file' field containing "
|
||||
"a file path (str), pathlib.Path, file-like object, or bytes"
|
||||
"a pathlib.Path, file-like object, or bytes"
|
||||
)
|
||||
|
||||
file_bytes: bytes
|
||||
mime_type: str = "application/octet-stream"
|
||||
file_name: Optional[str] = None
|
||||
|
||||
if isinstance(file_input, (str, Path)):
|
||||
if isinstance(file_input, str):
|
||||
# Bare strings are rejected here. The OCR ``document`` accepts a
|
||||
# ``{"type": "file", "file": <value>}`` shape, and when this helper
|
||||
# runs in a proxy request handler ``<value>`` is attacker-controlled.
|
||||
# Opening it as a path is an arbitrary local file read on the proxy
|
||||
# host, which is then base64-encoded and forwarded to the OCR
|
||||
# provider — an exfiltration primitive.
|
||||
raise ValueError(
|
||||
"OCR file input does not accept bare str values. Pass bytes, "
|
||||
"a pathlib.Path, or a file-like object. To OCR a local file "
|
||||
"from a path, call open(path, 'rb') yourself."
|
||||
)
|
||||
if isinstance(file_input, os.PathLike):
|
||||
# os.PathLike (pathlib.Path and custom __fspath__ classes) is a
|
||||
# Python-level type that HTTP form values can't fabricate.
|
||||
file_path = str(file_input)
|
||||
if not os.path.isfile(file_path):
|
||||
raise FileNotFoundError(f"File not found: {file_path}")
|
||||
|
|
@ -417,7 +432,7 @@ def convert_file_document_to_url_document(document: Dict[str, Any]) -> Dict[str,
|
|||
else:
|
||||
raise ValueError(
|
||||
f"Unsupported file input type: {type(file_input)}. "
|
||||
"Expected str (file path), pathlib.Path, bytes, or a file-like object."
|
||||
"Expected pathlib.Path, bytes, or a file-like object."
|
||||
)
|
||||
|
||||
if not file_bytes:
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import re
|
||||
from typing import Dict, List, Optional, Set, Tuple, cast
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
|
@ -122,6 +123,24 @@ class MCPRequestHandler:
|
|||
# cannot be smuggled via query string, hostname, or a deeper URL segment.
|
||||
if request.url.path.startswith("/.well-known/"):
|
||||
validated_user_api_key_auth = UserAPIKeyAuth()
|
||||
elif (
|
||||
not litellm_api_key
|
||||
and MCPRequestHandler._target_servers_delegate_auth_to_upstream( # noqa: E501
|
||||
path=request.url.path, mcp_servers=mcp_servers
|
||||
)
|
||||
):
|
||||
# Operator opted this oauth2 server into upstream-delegated auth
|
||||
# (PKCE passthrough): skip LiteLLM API-key/SSO entirely so the
|
||||
# client authenticates directly with the upstream MCP server.
|
||||
# Fires ONLY when neither x-litellm-api-key nor Authorization is
|
||||
# present. If any LiteLLM key is supplied (primary or secondary
|
||||
# header), we fall through so user_id is resolved, spend/rate
|
||||
# limiting apply, and any stored OAuth token can be retrieved
|
||||
# and forwarded upstream. Gated by
|
||||
# _target_servers_delegate_auth_to_upstream, which only returns
|
||||
# True when EVERY target is auth_type=oauth2 AND has the
|
||||
# delegate_auth_to_upstream flag set — fails closed otherwise.
|
||||
validated_user_api_key_auth = UserAPIKeyAuth()
|
||||
elif has_explicit_litellm_key:
|
||||
# Explicit x-litellm-api-key provided - always validate normally
|
||||
validated_user_api_key_auth = await user_api_key_auth(
|
||||
|
|
@ -181,23 +200,62 @@ class MCPRequestHandler:
|
|||
@staticmethod
|
||||
def _extract_target_server_names_from_path(path: str) -> List[str]:
|
||||
"""
|
||||
Extract the target MCP server name from the standard MCP transport
|
||||
URL patterns: ``/mcp/{server_name}[/...]`` and
|
||||
Extract the target MCP server name(s) from the standard MCP transport
|
||||
URL patterns: ``/mcp/{server_name_or_csv}[/...]`` and
|
||||
``/{server_name}/mcp[/...]``. Returns ``[]`` for any other path so
|
||||
callers fail closed when the target cannot be resolved.
|
||||
|
||||
Mirrors the regex-based parser in ``server.py::_get_mcp_servers_in_path``
|
||||
so the names used for auth gating match the names used for downstream
|
||||
filtering. Without this alignment, an attacker could craft
|
||||
``/mcp/<delegated_server>/<garbage>`` so that auth treats the request
|
||||
as targeting the delegate server (bypassing LiteLLM auth) while
|
||||
downstream filtering sees a different (non-existent) target and falls
|
||||
back to the caller's full allowed-server set.
|
||||
|
||||
REST/admin endpoints, OAuth2 server endpoints
|
||||
(``/{server_name}/authorize``, ``/token`` etc.), and ``.well-known``
|
||||
discovery routes intentionally fall through — those flows do not need
|
||||
OAuth2 token passthrough. Clients aggregating multiple servers should
|
||||
use ``x-mcp-servers``, which takes precedence over path parsing.
|
||||
use ``x-mcp-servers`` on a path that does not encode a target.
|
||||
"""
|
||||
# ``/{server_name}/mcp[/...]`` form — single server. The literal
|
||||
# ``mcp`` must be the second segment (not the first, which would be
|
||||
# the ``/mcp/...`` form handled below). This branch must stay in sync
|
||||
# with ``server.py::_get_mcp_servers_in_path``, which also accepts the
|
||||
# un-rewritten form (some entry points may skip the
|
||||
# ``dynamic_mcp_route`` rewrite).
|
||||
segments = [s for s in path.split("/") if s]
|
||||
if len(segments) >= 2 and segments[0] == "mcp":
|
||||
return [segments[1]]
|
||||
if len(segments) >= 2 and segments[1] == "mcp":
|
||||
if len(segments) >= 2 and segments[1] == "mcp" and segments[0] != "mcp":
|
||||
return [segments[0]]
|
||||
return []
|
||||
|
||||
# ``/mcp/...`` form — server name(s) may contain a slash (e.g.
|
||||
# ``custom_solutions/user_123``) and may be a comma-separated list.
|
||||
# Use the same parsing logic as ``_get_mcp_servers_in_path`` so the
|
||||
# parsed names match downstream routing.
|
||||
mcp_path_match = re.match(r"^/mcp/([^?#]+)(?:\?.*)?(?:#.*)?$", path)
|
||||
if not mcp_path_match:
|
||||
return []
|
||||
servers_and_path = mcp_path_match.group(1)
|
||||
if not servers_and_path:
|
||||
return []
|
||||
|
||||
if "," in servers_and_path:
|
||||
# Comma-separated servers, possibly followed by a trailing path.
|
||||
path_match = re.search(r"/([^/,]+(?:/[^/,]+)*)$", servers_and_path)
|
||||
if path_match:
|
||||
servers_part = servers_and_path[: -(len(path_match.group(1)) + 1)]
|
||||
else:
|
||||
servers_part = servers_and_path
|
||||
return [s.strip() for s in servers_part.split(",") if s.strip()]
|
||||
|
||||
# Single-server case — server name may contain at most one slash.
|
||||
single_server_match = re.match(
|
||||
r"^([^/]+(?:/[^/]+)?)(?:/.*)?$", servers_and_path
|
||||
)
|
||||
if single_server_match:
|
||||
return [single_server_match.group(1)]
|
||||
return [servers_and_path]
|
||||
|
||||
@staticmethod
|
||||
def _target_servers_use_oauth2(path: str, mcp_servers: Optional[List[str]]) -> bool:
|
||||
|
|
@ -217,13 +275,13 @@ class MCPRequestHandler:
|
|||
)
|
||||
from litellm.types.mcp import MCPAuth
|
||||
|
||||
# Use the x-mcp-servers header verbatim when present (including the
|
||||
# explicitly-empty list, which means "no targets" → fail closed).
|
||||
# Only fall back to path parsing when the header was absent entirely.
|
||||
target_names = (
|
||||
mcp_servers
|
||||
if mcp_servers is not None
|
||||
else MCPRequestHandler._extract_target_server_names_from_path(path)
|
||||
# Resolve the same target list downstream routing will use. For
|
||||
# ``/mcp/...`` routes, ``extract_mcp_auth_context`` overrides the
|
||||
# ``x-mcp-servers`` header with path-derived names, so we must mirror
|
||||
# that here — otherwise a caller could set the header to a permissive
|
||||
# server while the path targets a stricter one (header/path TOCTOU).
|
||||
target_names = MCPRequestHandler._resolve_target_server_names(
|
||||
path=path, mcp_servers_header=mcp_servers
|
||||
)
|
||||
if not target_names:
|
||||
return False
|
||||
|
|
@ -234,6 +292,78 @@ class MCPRequestHandler:
|
|||
return False
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def _target_servers_delegate_auth_to_upstream(
|
||||
path: str, mcp_servers: Optional[List[str]]
|
||||
) -> bool:
|
||||
"""
|
||||
True only when EVERY MCP server the request targets is configured for
|
||||
``auth_type == oauth2`` AND has ``delegate_auth_to_upstream=True``.
|
||||
Fails closed when any target does not opt in or cannot be resolved.
|
||||
|
||||
Used by :meth:`process_mcp_request` to skip LiteLLM API-key/SSO auth
|
||||
entirely (PKCE passthrough) so the client authenticates directly with
|
||||
the upstream MCP server. Mixed-target requests (e.g. one delegated +
|
||||
one non-delegated server) fall back to normal LiteLLM auth.
|
||||
"""
|
||||
# Inline imports avoid a circular dependency: mcp_server_manager imports
|
||||
# from this module.
|
||||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import (
|
||||
global_mcp_server_manager,
|
||||
)
|
||||
from litellm.types.mcp import MCPAuth
|
||||
|
||||
# See _target_servers_use_oauth2: must mirror the downstream
|
||||
# header-vs-path override or an attacker could set
|
||||
# ``x-mcp-servers`` to a delegate-enabled server while the URL path
|
||||
# targets a non-delegate server, skipping LiteLLM auth for it.
|
||||
target_names = MCPRequestHandler._resolve_target_server_names(
|
||||
path=path, mcp_servers_header=mcp_servers
|
||||
)
|
||||
if not target_names:
|
||||
return False
|
||||
|
||||
for name in target_names:
|
||||
server = global_mcp_server_manager.get_mcp_server_by_name(name)
|
||||
if server is None or server.auth_type != MCPAuth.oauth2:
|
||||
return False
|
||||
# `is True` is intentional: opt-in must be an explicit boolean
|
||||
# True. A MagicMock attribute (in tests) or any other truthy
|
||||
# non-bool must not silently enable the bypass.
|
||||
if getattr(server, "delegate_auth_to_upstream", False) is not True:
|
||||
return False
|
||||
if not getattr(server, "available_on_public_internet", True):
|
||||
return False
|
||||
# Never delegate for M2M (client_credentials) servers: LiteLLM
|
||||
# fetches the upstream token automatically using stored credentials,
|
||||
# so allowing anonymous bypass would let any external caller invoke
|
||||
# tools authenticated as LiteLLM's service account.
|
||||
if server.has_client_credentials:
|
||||
return False
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def _resolve_target_server_names(
|
||||
path: str, mcp_servers_header: Optional[List[str]]
|
||||
) -> List[str]:
|
||||
"""
|
||||
Resolve the target MCP server names exactly as downstream routing
|
||||
does (``server.py::extract_mcp_auth_context``).
|
||||
|
||||
For ``/mcp/...`` paths, downstream routing **overrides** any
|
||||
``x-mcp-servers`` header value with the path-derived names. Mirror
|
||||
that here so an attacker cannot use a permissive header value to
|
||||
flip an auth gate while the path targets a stricter server
|
||||
(header/path TOCTOU). For non-``/mcp/...`` paths (where the path
|
||||
does not encode targets), fall back to the header.
|
||||
"""
|
||||
path_targets = MCPRequestHandler._extract_target_server_names_from_path(path)
|
||||
if path_targets:
|
||||
return path_targets
|
||||
# Path did not resolve to /mcp/... targets — trust the header
|
||||
# (including an explicitly empty list, which means "no targets").
|
||||
return mcp_servers_header if mcp_servers_header is not None else []
|
||||
|
||||
@staticmethod
|
||||
def _get_mcp_auth_header_from_headers(headers: Headers) -> Optional[str]:
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -402,6 +402,9 @@ class MCPServerManager:
|
|||
available_on_public_internet=bool(
|
||||
server_config.get("available_on_public_internet", True)
|
||||
),
|
||||
delegate_auth_to_upstream=bool(
|
||||
server_config.get("delegate_auth_to_upstream", False)
|
||||
),
|
||||
# AWS SigV4 fields
|
||||
aws_access_key_id=server_config.get("aws_access_key_id", None),
|
||||
aws_secret_access_key=server_config.get("aws_secret_access_key", None),
|
||||
|
|
@ -599,16 +602,57 @@ class MCPServerManager:
|
|||
)
|
||||
raise e
|
||||
|
||||
def _cleanup_server_tool_routing_artifacts(self, server: MCPServer) -> None:
|
||||
"""Drop OpenAPI global tools and name-mapping rows owned by ``server``.
|
||||
|
||||
When a server leaves ``self.registry`` (eviction, ``remove_server``, etc.),
|
||||
OpenAPI tools remain in ``global_mcp_tool_registry`` and
|
||||
``tool_name_to_mcp_server_name_mapping`` unless removed here. Stale
|
||||
mappings make ``_get_mcp_server_from_tool_name`` resolve to a prefix that
|
||||
no longer exists in the live registry.
|
||||
"""
|
||||
from litellm.proxy._experimental.mcp_server.tool_registry import (
|
||||
global_mcp_tool_registry,
|
||||
)
|
||||
|
||||
prefix_root = normalize_server_name(get_server_prefix(server))
|
||||
if server.spec_path and prefix_root:
|
||||
openapi_key_prefix = prefix_root + MCP_TOOL_PREFIX_SEPARATOR
|
||||
global_mcp_tool_registry.unregister_tools_with_prefix(openapi_key_prefix)
|
||||
|
||||
owned_raw: Set[str] = set()
|
||||
for p in iter_known_server_prefixes(server):
|
||||
if p:
|
||||
owned_raw.add(p)
|
||||
if server.name:
|
||||
owned_raw.add(server.name)
|
||||
|
||||
owned_normalized = {normalize_server_name(x) for x in owned_raw}
|
||||
|
||||
stale_mapping_keys: List[str] = []
|
||||
for tool_name, mapped_server in list(
|
||||
self.tool_name_to_mcp_server_name_mapping.items()
|
||||
):
|
||||
if mapped_server in owned_raw:
|
||||
stale_mapping_keys.append(tool_name)
|
||||
elif normalize_server_name(str(mapped_server)) in owned_normalized:
|
||||
stale_mapping_keys.append(tool_name)
|
||||
|
||||
for key in stale_mapping_keys:
|
||||
del self.tool_name_to_mcp_server_name_mapping[key]
|
||||
|
||||
def remove_server(self, mcp_server: LiteLLM_MCPServerTable):
|
||||
"""
|
||||
Remove a server from the registry
|
||||
"""
|
||||
if mcp_server.server_name in self.get_registry():
|
||||
del self.registry[mcp_server.server_name]
|
||||
verbose_logger.debug(f"Removed MCP Server: {mcp_server.server_name}")
|
||||
elif mcp_server.server_id in self.get_registry():
|
||||
del self.registry[mcp_server.server_id]
|
||||
verbose_logger.debug(f"Removed MCP Server: {mcp_server.server_id}")
|
||||
evicted: Optional[MCPServer] = self.registry.pop(mcp_server.server_id, None)
|
||||
if evicted is None and mcp_server.server_name:
|
||||
evicted = self.registry.pop(mcp_server.server_name, None)
|
||||
if evicted is not None:
|
||||
verbose_logger.debug(
|
||||
"Removed MCP Server: %s", mcp_server.server_id or mcp_server.server_name
|
||||
)
|
||||
self._cleanup_server_tool_routing_artifacts(evicted)
|
||||
else:
|
||||
verbose_logger.warning(
|
||||
f"Server ID {mcp_server.server_id} not found in registry"
|
||||
|
|
@ -755,6 +799,9 @@ class MCPServerManager:
|
|||
available_on_public_internet=bool(
|
||||
getattr(mcp_server, "available_on_public_internet", True)
|
||||
),
|
||||
delegate_auth_to_upstream=bool(
|
||||
getattr(mcp_server, "delegate_auth_to_upstream", False)
|
||||
),
|
||||
created_at=getattr(mcp_server, "created_at", None),
|
||||
updated_at=getattr(mcp_server, "updated_at", None),
|
||||
tool_name_to_display_name=_deserialize_json_dict(
|
||||
|
|
@ -806,6 +853,13 @@ class MCPServerManager:
|
|||
self.initialize_tool_name_to_mcp_server_name_mapping()
|
||||
|
||||
async def add_server(self, mcp_server: LiteLLM_MCPServerTable):
|
||||
# The runtime registry is the allowlist for tool calls and health
|
||||
# probes (which spawn the underlying transport, including stdio
|
||||
# subprocesses). Match the eligibility set used by the bulk DB
|
||||
# filter in reload_servers_from_database() — NULL is legacy and
|
||||
# "approved" is a legacy alias for "active".
|
||||
if mcp_server.approval_status not in (None, "active", "approved"):
|
||||
return
|
||||
try:
|
||||
if mcp_server.server_id not in self.registry:
|
||||
new_server = await self.build_mcp_server_from_table(mcp_server)
|
||||
|
|
@ -819,6 +873,16 @@ class MCPServerManager:
|
|||
raise e
|
||||
|
||||
async def update_server(self, mcp_server: LiteLLM_MCPServerTable):
|
||||
# If a previously-active server has been moved out of the active
|
||||
# state, evict any stale registry entry so subsequent tool calls and
|
||||
# health probes can't reach it.
|
||||
if mcp_server.approval_status not in (None, "active", "approved"):
|
||||
evicted = self.registry.pop(mcp_server.server_id, None)
|
||||
if evicted is None and mcp_server.server_name:
|
||||
evicted = self.registry.pop(mcp_server.server_name, None)
|
||||
if evicted is not None:
|
||||
self._cleanup_server_tool_routing_artifacts(evicted)
|
||||
return
|
||||
try:
|
||||
if mcp_server.server_id in self.registry:
|
||||
new_server = await self.build_mcp_server_from_table(mcp_server)
|
||||
|
|
@ -909,6 +973,34 @@ class MCPServerManager:
|
|||
if not in_toolset_scope:
|
||||
combined_servers.update(allow_all_server_ids)
|
||||
|
||||
# For anonymous callers (no user_id, no role), also surface any
|
||||
# servers the operator has opted into upstream-delegated auth.
|
||||
# These servers handle their own auth at the upstream level, so
|
||||
# LiteLLM granting access here does not bypass any security gate.
|
||||
is_anonymous = not (
|
||||
user_api_key_auth
|
||||
and (
|
||||
getattr(user_api_key_auth, "user_id", None)
|
||||
or getattr(user_api_key_auth, "user_role", None)
|
||||
or getattr(user_api_key_auth, "api_key", None)
|
||||
)
|
||||
)
|
||||
if is_anonymous:
|
||||
delegate_server_ids = [
|
||||
server.server_id
|
||||
for server in self.get_registry().values()
|
||||
if getattr(server, "auth_type", None) == MCPAuth.oauth2
|
||||
and getattr(server, "delegate_auth_to_upstream", False) is True
|
||||
# M2M servers must not be exposed anonymously: an
|
||||
# unauthenticated caller would get LiteLLM to proxy tool
|
||||
# calls using its stored client_credentials.
|
||||
and not server.has_client_credentials
|
||||
# Internal-only servers must not be reachable from public
|
||||
# internet callers who happen to carry an upstream token.
|
||||
and getattr(server, "available_on_public_internet", True)
|
||||
]
|
||||
combined_servers.update(delegate_server_ids)
|
||||
|
||||
if len(combined_servers) == 0:
|
||||
verbose_logger.debug(
|
||||
"No allowed MCP Servers found for user api key auth."
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ from typing import (
|
|||
cast,
|
||||
)
|
||||
|
||||
import httpx
|
||||
from fastapi import FastAPI, HTTPException
|
||||
from pydantic import AnyUrl, ConfigDict
|
||||
from starlette.requests import Request as StarletteRequest
|
||||
|
|
@ -51,13 +52,17 @@ from litellm.proxy._experimental.mcp_server.utils import (
|
|||
get_server_prefix,
|
||||
iter_known_server_prefixes,
|
||||
)
|
||||
from litellm.llms.custom_httpx.http_handler import (
|
||||
get_async_httpx_client,
|
||||
httpxSpecialProvider,
|
||||
)
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.ip_address_utils import IPAddressUtils
|
||||
from litellm.proxy.litellm_pre_call_utils import (
|
||||
LiteLLMProxyRequestSetup,
|
||||
get_chain_id_from_headers,
|
||||
)
|
||||
from litellm.types.mcp import MCPAuth
|
||||
from litellm.types.mcp import MCPAuth, MCPSpecVersion
|
||||
from litellm.types.mcp_server.mcp_server_manager import MCPInfo, MCPServer
|
||||
from litellm.types.utils import CallTypes, StandardLoggingMCPToolCall
|
||||
from litellm.utils import Rules, client, function_setup
|
||||
|
|
@ -1332,8 +1337,24 @@ if MCP_AVAILABLE:
|
|||
raw_headers=raw_headers,
|
||||
)
|
||||
|
||||
# If no OAuth2 token came from request headers, fall back to pre-fetched creds
|
||||
if extra_headers is None and server.auth_type == MCPAuth.oauth2:
|
||||
# Prefer server-stored per-user OAuth when configured, so a stale
|
||||
# Authorization header from the MCP client cannot override Redis/DB
|
||||
# (same issue as call_tool in mcp_server_manager: VS Code caches tokens).
|
||||
if (
|
||||
server.auth_type == MCPAuth.oauth2
|
||||
and getattr(server, "needs_user_oauth_token", False)
|
||||
and user_api_key_auth is not None
|
||||
):
|
||||
db_headers = await _get_user_oauth_extra_headers_from_db(
|
||||
server,
|
||||
user_api_key_auth,
|
||||
prefetched_creds=_prefetched_oauth_creds,
|
||||
)
|
||||
if db_headers:
|
||||
extra_headers = db_headers
|
||||
|
||||
# If still no OAuth2 token, fall back to pre-fetched creds (non-stale-client path)
|
||||
elif extra_headers is None and server.auth_type == MCPAuth.oauth2:
|
||||
extra_headers = await _get_user_oauth_extra_headers_from_db(
|
||||
server,
|
||||
user_api_key_auth,
|
||||
|
|
@ -2536,6 +2557,10 @@ if MCP_AVAILABLE:
|
|||
import re
|
||||
|
||||
mcp_servers_from_path: Optional[List[str]] = None
|
||||
segments = [s for s in path.split("/") if s]
|
||||
if len(segments) >= 2 and segments[1] == "mcp" and segments[0] != "mcp":
|
||||
return [segments[0]]
|
||||
|
||||
# Match /mcp/<servers_and_maybe_path>
|
||||
# Where servers can be comma-separated list of server names
|
||||
# Server names can contain slashes (e.g., "custom_solutions/user_123")
|
||||
|
|
@ -2754,6 +2779,157 @@ if MCP_AVAILABLE:
|
|||
)
|
||||
return user_api_key_auth.model_copy(update={"object_permission": updated_op})
|
||||
|
||||
def _get_forwarded_auth_from_scope(scope: Scope) -> Optional[str]:
|
||||
"""Return the upstream-bound ``Authorization`` header value, or None.
|
||||
|
||||
Only returns the ``Authorization`` header when ``x-litellm-api-key`` is
|
||||
also present. In that case ``Authorization`` is unambiguously the
|
||||
upstream token the caller wants forwarded to the MCP server. When
|
||||
``x-litellm-api-key`` is absent the ``Authorization`` header may itself
|
||||
be the LiteLLM proxy API key (backward-compat path in
|
||||
``MCPRequestHandler.process_mcp_request``), and forwarding it upstream
|
||||
would leak the proxy key to a third-party MCP server.
|
||||
"""
|
||||
authorization = None
|
||||
has_litellm_key_header = False
|
||||
for key, value in scope.get("headers", []):
|
||||
key_lower = key.lower()
|
||||
if key_lower == b"authorization":
|
||||
authorization = value.decode("latin-1")
|
||||
elif key_lower == b"x-litellm-api-key":
|
||||
has_litellm_key_header = True
|
||||
if not has_litellm_key_header:
|
||||
return None
|
||||
return authorization
|
||||
|
||||
async def _probe_upstream_auth(
|
||||
url: str,
|
||||
auth_header: str,
|
||||
timeout: float = 5.0,
|
||||
) -> tuple:
|
||||
"""JSON-RPC initialize-probe the upstream URL to check whether the token is accepted.
|
||||
|
||||
Uses POST so StreamableHTTP MCP servers run the same auth path as a
|
||||
real client request. Returns (status_code, www_authenticate).
|
||||
Fails-open with (200, None) on network errors so a transient hiccup
|
||||
does not block valid requests.
|
||||
|
||||
Uses the public ``AsyncHTTPHandler.post()`` interface and catches
|
||||
``httpx.HTTPStatusError`` separately so the 401/403 we want to surface
|
||||
is not swallowed by the broad fail-open ``except Exception`` below.
|
||||
"""
|
||||
client = get_async_httpx_client(
|
||||
llm_provider=httpxSpecialProvider.MCP,
|
||||
params={"timeout": timeout},
|
||||
)
|
||||
probe_payload = {
|
||||
"jsonrpc": "2.0",
|
||||
"id": "litellm-mcp-auth-probe",
|
||||
"method": "initialize",
|
||||
"params": {
|
||||
"protocolVersion": MCPSpecVersion.jun_2025.value,
|
||||
"capabilities": {},
|
||||
"clientInfo": {
|
||||
"name": "litellm-mcp-auth-probe",
|
||||
"version": "1.0.0",
|
||||
},
|
||||
},
|
||||
}
|
||||
probe_headers = {
|
||||
"Authorization": auth_header,
|
||||
"Accept": "application/json, text/event-stream",
|
||||
}
|
||||
try:
|
||||
resp = await client.post(
|
||||
url=url,
|
||||
headers=probe_headers,
|
||||
json=probe_payload,
|
||||
timeout=timeout,
|
||||
)
|
||||
return resp.status_code, resp.headers.get("www-authenticate")
|
||||
except httpx.HTTPStatusError as exc:
|
||||
# AsyncHTTPHandler.post() calls raise_for_status(); a 401/403 from
|
||||
# upstream lands here. Return its status so the caller can map it
|
||||
# to the appropriate response.
|
||||
return exc.response.status_code, exc.response.headers.get(
|
||||
"www-authenticate"
|
||||
)
|
||||
except Exception as exc:
|
||||
verbose_logger.debug(
|
||||
f"_probe_upstream_auth: probe to {url} failed ({exc}), allowing request through"
|
||||
)
|
||||
return 200, None
|
||||
|
||||
async def _check_passthrough_upstream_auth(
|
||||
scope: Scope,
|
||||
user_api_key_auth: Optional[UserAPIKeyAuth],
|
||||
mcp_servers: Optional[List[str]],
|
||||
client_ip: Optional[str],
|
||||
) -> None:
|
||||
"""Probe pass-through upstream servers in parallel before the MCP session starts.
|
||||
|
||||
Only servers the caller's key is already authorized to reach are probed —
|
||||
the list is derived from _get_allowed_mcp_servers so that a user cannot
|
||||
trigger an upstream probe against a server their key is not permitted for.
|
||||
|
||||
The MCP SDK commits HTTP 200 headers before invoking handlers, so a 401
|
||||
can only be returned before that point. This function raises HTTPException(401)
|
||||
with a WWW-Authenticate header if any upstream rejects the client token.
|
||||
Fails-open: network errors are logged and the request is allowed through.
|
||||
"""
|
||||
forwarded_auth = _get_forwarded_auth_from_scope(scope)
|
||||
if not forwarded_auth:
|
||||
return
|
||||
|
||||
# Use the authorized server set, not the raw user-supplied names, so that
|
||||
# a caller cannot force a probe to a server their key is not allowed to use.
|
||||
allowed_servers = await _get_allowed_mcp_servers(
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
mcp_servers=mcp_servers,
|
||||
client_ip=client_ip,
|
||||
)
|
||||
passthrough_servers = [
|
||||
srv
|
||||
for srv in allowed_servers
|
||||
if srv.extra_headers
|
||||
and any(h.lower() == "authorization" for h in srv.extra_headers)
|
||||
# Exclude M2M servers: _prepare_mcp_server_headers skips caller
|
||||
# Authorization when has_client_credentials is set, so probing
|
||||
# those with the caller's token would send the wrong credential.
|
||||
and not srv.has_client_credentials
|
||||
]
|
||||
if not passthrough_servers:
|
||||
return
|
||||
|
||||
probe_results = await asyncio.gather(
|
||||
*[
|
||||
_probe_upstream_auth(srv.url or "", forwarded_auth)
|
||||
for srv in passthrough_servers
|
||||
]
|
||||
)
|
||||
request = StarletteRequest(scope)
|
||||
base_url = get_request_base_url(request)
|
||||
for srv, (probe_status, _) in zip(passthrough_servers, probe_results):
|
||||
if probe_status == 401:
|
||||
# Token is missing or expired — direct the client to re-authorize.
|
||||
authorization_uri = (
|
||||
f"Bearer authorization_uri="
|
||||
f"{base_url}/.well-known/oauth-authorization-server/{srv.name}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail="Unauthorized",
|
||||
headers={"WWW-Authenticate": authorization_uri},
|
||||
)
|
||||
if probe_status == 403:
|
||||
# Token is valid but the caller lacks permission — do not hint
|
||||
# at re-authorization (RFC 9110: a fresh token with the same
|
||||
# scopes would just hit 403 again and loop indefinitely).
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail="Forbidden",
|
||||
)
|
||||
|
||||
async def handle_streamable_http_mcp(
|
||||
scope: Scope, receive: Receive, send: Send
|
||||
) -> None:
|
||||
|
|
@ -2827,6 +3003,13 @@ if MCP_AVAILABLE:
|
|||
user_api_key_auth, active_toolset_id
|
||||
)
|
||||
|
||||
# Pre-flight auth check for pass-through servers. Must run after
|
||||
# toolset scoping so the probe list is derived from the fully-authorized
|
||||
# server set, not the raw user-supplied names.
|
||||
await _check_passthrough_upstream_auth(
|
||||
scope, user_api_key_auth, mcp_servers, _client_ip
|
||||
)
|
||||
|
||||
# Inject masked debug headers when client sends x-litellm-mcp-debug: true
|
||||
_debug_headers = MCPDebug.maybe_build_debug_headers(
|
||||
raw_headers=raw_headers,
|
||||
|
|
|
|||
|
|
@ -59,6 +59,22 @@ class MCPToolRegistry:
|
|||
]
|
||||
return list(self.tools.values())
|
||||
|
||||
def unregister_tools_with_prefix(self, prefix: str) -> int:
|
||||
"""Remove tools whose registered name starts with ``prefix``.
|
||||
|
||||
Used when an OpenAPI-backed MCP server leaves the runtime registry so
|
||||
stale tool handlers cannot be invoked after eviction.
|
||||
"""
|
||||
if not prefix:
|
||||
return 0
|
||||
removed = 0
|
||||
for name in list(self.tools.keys()):
|
||||
if name.startswith(prefix):
|
||||
del self.tools[name]
|
||||
removed += 1
|
||||
verbose_logger.debug("Unregistered MCP tool %s", name)
|
||||
return removed
|
||||
|
||||
def convert_tools_to_mcp_sdk_tool_type(
|
||||
self, tools: List[MCPTool]
|
||||
) -> List["MCPToolSDKTool"]:
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,30 +1,30 @@
|
|||
1:"$Sreact.fragment"
|
||||
2:I[347257,["/litellm-asset-prefix/_next/static/chunks/d96012bcfc98706a.js","/litellm-asset-prefix/_next/static/chunks/dbca964212122d58.js"],"ClientPageRoot"]
|
||||
3:I[952683,["/litellm-asset-prefix/_next/static/chunks/9e09de50158b3159.js","/litellm-asset-prefix/_next/static/chunks/7e5fe5584502da06.js","/litellm-asset-prefix/_next/static/chunks/0493aafc4891dd29.js","/litellm-asset-prefix/_next/static/chunks/f7e1d08418645368.js","/litellm-asset-prefix/_next/static/chunks/b3d198d6c56a21b8.js","/litellm-asset-prefix/_next/static/chunks/403c4d96324c23a6.js","/litellm-asset-prefix/_next/static/chunks/37e77c06e99eb8ff.js","/litellm-asset-prefix/_next/static/chunks/adb8beb738574863.js","/litellm-asset-prefix/_next/static/chunks/0549bc9afa7d4888.js","/litellm-asset-prefix/_next/static/chunks/0b470ffc60999bf4.js","/litellm-asset-prefix/_next/static/chunks/c847ecdf8c790b0b.js","/litellm-asset-prefix/_next/static/chunks/baadbd26839e7b66.js","/litellm-asset-prefix/_next/static/chunks/ee5f9a39a526e423.js","/litellm-asset-prefix/_next/static/chunks/6eee262391715440.js","/litellm-asset-prefix/_next/static/chunks/4e17b625d75327a7.js","/litellm-asset-prefix/_next/static/chunks/7b788dd93ad868b3.js","/litellm-asset-prefix/_next/static/chunks/a06cc76a774dd182.js","/litellm-asset-prefix/_next/static/chunks/264fd32eefec52b6.js","/litellm-asset-prefix/_next/static/chunks/86828bdbafb8b581.js","/litellm-asset-prefix/_next/static/chunks/10dc4591ef08a91f.js","/litellm-asset-prefix/_next/static/chunks/e099566e8bd4ee4e.js","/litellm-asset-prefix/_next/static/chunks/fbe12a36d22e9554.js","/litellm-asset-prefix/_next/static/chunks/2971c4658f1bcd7d.js","/litellm-asset-prefix/_next/static/chunks/134f728fa7099e3e.js","/litellm-asset-prefix/_next/static/chunks/679dbd657c8b5aef.js","/litellm-asset-prefix/_next/static/chunks/94f7208f5087e27c.js","/litellm-asset-prefix/_next/static/chunks/43f6fc3c2ab9cf23.js","/litellm-asset-prefix/_next/static/chunks/4e06277331e725da.js","/litellm-asset-prefix/_next/static/chunks/3b30ab8eaa03bc21.js","/litellm-asset-prefix/_next/static/chunks/ac3cf77acb5bf234.js","/litellm-asset-prefix/_next/static/chunks/fb125648f2dae104.js","/litellm-asset-prefix/_next/static/chunks/7e417dd24c8becd0.js","/litellm-asset-prefix/_next/static/chunks/a09028cd611c08ef.js","/litellm-asset-prefix/_next/static/chunks/6967a3b4ecbd3785.js","/litellm-asset-prefix/_next/static/chunks/3e917c79aadd945b.js","/litellm-asset-prefix/_next/static/chunks/9bbebdeb3f1cb03f.js","/litellm-asset-prefix/_next/static/chunks/0a65da2cd24e2ab6.js","/litellm-asset-prefix/_next/static/chunks/908828a91f602d8b.js","/litellm-asset-prefix/_next/static/chunks/5f2d62a75803a3f7.js","/litellm-asset-prefix/_next/static/chunks/ca5fbafaf3826374.js","/litellm-asset-prefix/_next/static/chunks/d3ac82723ec9e30d.js","/litellm-asset-prefix/_next/static/chunks/9b0ee76cbdef1a2a.js","/litellm-asset-prefix/_next/static/chunks/1bc2898be56acd1b.js","/litellm-asset-prefix/_next/static/chunks/fcdf7322b0aa3e2e.js","/litellm-asset-prefix/_next/static/chunks/878832edb30e99a4.js","/litellm-asset-prefix/_next/static/chunks/496b84010c33cf69.js","/litellm-asset-prefix/_next/static/chunks/8e3d0ce9505a304f.js","/litellm-asset-prefix/_next/static/chunks/e1f23fd814ac3500.js","/litellm-asset-prefix/_next/static/chunks/88c74f8b4b20d25a.js","/litellm-asset-prefix/_next/static/chunks/99cf9cf99df5ccfc.js","/litellm-asset-prefix/_next/static/chunks/4980372eaa37b78b.js","/litellm-asset-prefix/_next/static/chunks/0cdfadbcf4b8c9e4.js","/litellm-asset-prefix/_next/static/chunks/8f3bf592254c6c3b.js","/litellm-asset-prefix/_next/static/chunks/8c17e934bd227606.js","/litellm-asset-prefix/_next/static/chunks/b98447395b5d37ef.js"],"default"]
|
||||
3:I[952683,["/litellm-asset-prefix/_next/static/chunks/9e09de50158b3159.js","/litellm-asset-prefix/_next/static/chunks/7e5fe5584502da06.js","/litellm-asset-prefix/_next/static/chunks/0493aafc4891dd29.js","/litellm-asset-prefix/_next/static/chunks/f7e1d08418645368.js","/litellm-asset-prefix/_next/static/chunks/b3d198d6c56a21b8.js","/litellm-asset-prefix/_next/static/chunks/403c4d96324c23a6.js","/litellm-asset-prefix/_next/static/chunks/1d1c8edf97a801b6.js","/litellm-asset-prefix/_next/static/chunks/adb8beb738574863.js","/litellm-asset-prefix/_next/static/chunks/0549bc9afa7d4888.js","/litellm-asset-prefix/_next/static/chunks/0b470ffc60999bf4.js","/litellm-asset-prefix/_next/static/chunks/c847ecdf8c790b0b.js","/litellm-asset-prefix/_next/static/chunks/e099566e8bd4ee4e.js","/litellm-asset-prefix/_next/static/chunks/ee5f9a39a526e423.js","/litellm-asset-prefix/_next/static/chunks/b1c98cc932a0ab19.js","/litellm-asset-prefix/_next/static/chunks/4e17b625d75327a7.js","/litellm-asset-prefix/_next/static/chunks/7b788dd93ad868b3.js","/litellm-asset-prefix/_next/static/chunks/a06cc76a774dd182.js","/litellm-asset-prefix/_next/static/chunks/ca5fbafaf3826374.js","/litellm-asset-prefix/_next/static/chunks/7caea73b77a79d3c.js","/litellm-asset-prefix/_next/static/chunks/10dc4591ef08a91f.js","/litellm-asset-prefix/_next/static/chunks/0b3d09ff6c6e4335.js","/litellm-asset-prefix/_next/static/chunks/baadbd26839e7b66.js","/litellm-asset-prefix/_next/static/chunks/2971c4658f1bcd7d.js","/litellm-asset-prefix/_next/static/chunks/134f728fa7099e3e.js","/litellm-asset-prefix/_next/static/chunks/679dbd657c8b5aef.js","/litellm-asset-prefix/_next/static/chunks/94f7208f5087e27c.js","/litellm-asset-prefix/_next/static/chunks/43f6fc3c2ab9cf23.js","/litellm-asset-prefix/_next/static/chunks/4e06277331e725da.js","/litellm-asset-prefix/_next/static/chunks/3b30ab8eaa03bc21.js","/litellm-asset-prefix/_next/static/chunks/7e417dd24c8becd0.js","/litellm-asset-prefix/_next/static/chunks/da1c7742cc6fe8b4.js","/litellm-asset-prefix/_next/static/chunks/ca7a3fdb635fb7dc.js","/litellm-asset-prefix/_next/static/chunks/a09028cd611c08ef.js","/litellm-asset-prefix/_next/static/chunks/77e1b16e6f85230c.js","/litellm-asset-prefix/_next/static/chunks/ad02f56c287539eb.js","/litellm-asset-prefix/_next/static/chunks/908828a91f602d8b.js","/litellm-asset-prefix/_next/static/chunks/0a65da2cd24e2ab6.js","/litellm-asset-prefix/_next/static/chunks/fcdf7322b0aa3e2e.js","/litellm-asset-prefix/_next/static/chunks/a8f7c8c5eeb6e042.js","/litellm-asset-prefix/_next/static/chunks/4980372eaa37b78b.js","/litellm-asset-prefix/_next/static/chunks/d3ac82723ec9e30d.js","/litellm-asset-prefix/_next/static/chunks/496b84010c33cf69.js","/litellm-asset-prefix/_next/static/chunks/1bc2898be56acd1b.js","/litellm-asset-prefix/_next/static/chunks/6188170a32c9a3c3.js","/litellm-asset-prefix/_next/static/chunks/878832edb30e99a4.js","/litellm-asset-prefix/_next/static/chunks/934dbc43f8c1abde.js","/litellm-asset-prefix/_next/static/chunks/a0f7bfbaffe81a17.js","/litellm-asset-prefix/_next/static/chunks/e1f23fd814ac3500.js","/litellm-asset-prefix/_next/static/chunks/88c74f8b4b20d25a.js","/litellm-asset-prefix/_next/static/chunks/b948aa17e97c458d.js","/litellm-asset-prefix/_next/static/chunks/20acf4fa815c638e.js","/litellm-asset-prefix/_next/static/chunks/659ce28f2cb74401.js","/litellm-asset-prefix/_next/static/chunks/d6ab357d1bbb53f0.js","/litellm-asset-prefix/_next/static/chunks/99cf9cf99df5ccfc.js","/litellm-asset-prefix/_next/static/chunks/b3d631e60d6e8e9b.js"],"default"]
|
||||
1a:I[897367,["/litellm-asset-prefix/_next/static/chunks/d96012bcfc98706a.js","/litellm-asset-prefix/_next/static/chunks/dbca964212122d58.js"],"OutletBoundary"]
|
||||
1b:"$Sreact.suspense"
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/3f3fa56b5786d58c.css","style"]
|
||||
0:{"buildId":"8TZ2JbOi7SZ6BCj9ScTHW","rsc":["$","$1","c",{"children":[["$","$L2",null,{"Component":"$3","serverProvidedParams":{"searchParams":{},"params":{},"promises":["$@4","$@5"]}}],[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/3f3fa56b5786d58c.css","precedence":"next"}],["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/0493aafc4891dd29.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/f7e1d08418645368.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/b3d198d6c56a21b8.js","async":true}],["$","script","script-3",{"src":"/litellm-asset-prefix/_next/static/chunks/403c4d96324c23a6.js","async":true}],["$","script","script-4",{"src":"/litellm-asset-prefix/_next/static/chunks/37e77c06e99eb8ff.js","async":true}],["$","script","script-5",{"src":"/litellm-asset-prefix/_next/static/chunks/adb8beb738574863.js","async":true}],["$","script","script-6",{"src":"/litellm-asset-prefix/_next/static/chunks/0549bc9afa7d4888.js","async":true}],["$","script","script-7",{"src":"/litellm-asset-prefix/_next/static/chunks/0b470ffc60999bf4.js","async":true}],["$","script","script-8",{"src":"/litellm-asset-prefix/_next/static/chunks/c847ecdf8c790b0b.js","async":true}],["$","script","script-9",{"src":"/litellm-asset-prefix/_next/static/chunks/baadbd26839e7b66.js","async":true}],["$","script","script-10",{"src":"/litellm-asset-prefix/_next/static/chunks/ee5f9a39a526e423.js","async":true}],["$","script","script-11",{"src":"/litellm-asset-prefix/_next/static/chunks/6eee262391715440.js","async":true}],["$","script","script-12",{"src":"/litellm-asset-prefix/_next/static/chunks/4e17b625d75327a7.js","async":true}],["$","script","script-13",{"src":"/litellm-asset-prefix/_next/static/chunks/7b788dd93ad868b3.js","async":true}],["$","script","script-14",{"src":"/litellm-asset-prefix/_next/static/chunks/a06cc76a774dd182.js","async":true}],["$","script","script-15",{"src":"/litellm-asset-prefix/_next/static/chunks/264fd32eefec52b6.js","async":true}],["$","script","script-16",{"src":"/litellm-asset-prefix/_next/static/chunks/86828bdbafb8b581.js","async":true}],["$","script","script-17",{"src":"/litellm-asset-prefix/_next/static/chunks/10dc4591ef08a91f.js","async":true}],["$","script","script-18",{"src":"/litellm-asset-prefix/_next/static/chunks/e099566e8bd4ee4e.js","async":true}],["$","script","script-19",{"src":"/litellm-asset-prefix/_next/static/chunks/fbe12a36d22e9554.js","async":true}],["$","script","script-20",{"src":"/litellm-asset-prefix/_next/static/chunks/2971c4658f1bcd7d.js","async":true}],["$","script","script-21",{"src":"/litellm-asset-prefix/_next/static/chunks/134f728fa7099e3e.js","async":true}],["$","script","script-22",{"src":"/litellm-asset-prefix/_next/static/chunks/679dbd657c8b5aef.js","async":true}],["$","script","script-23",{"src":"/litellm-asset-prefix/_next/static/chunks/94f7208f5087e27c.js","async":true}],["$","script","script-24",{"src":"/litellm-asset-prefix/_next/static/chunks/43f6fc3c2ab9cf23.js","async":true}],["$","script","script-25",{"src":"/litellm-asset-prefix/_next/static/chunks/4e06277331e725da.js","async":true}],["$","script","script-26",{"src":"/litellm-asset-prefix/_next/static/chunks/3b30ab8eaa03bc21.js","async":true}],["$","script","script-27",{"src":"/litellm-asset-prefix/_next/static/chunks/ac3cf77acb5bf234.js","async":true}],["$","script","script-28",{"src":"/litellm-asset-prefix/_next/static/chunks/fb125648f2dae104.js","async":true}],["$","script","script-29",{"src":"/litellm-asset-prefix/_next/static/chunks/7e417dd24c8becd0.js","async":true}],["$","script","script-30",{"src":"/litellm-asset-prefix/_next/static/chunks/a09028cd611c08ef.js","async":true}],["$","script","script-31",{"src":"/litellm-asset-prefix/_next/static/chunks/6967a3b4ecbd3785.js","async":true}],["$","script","script-32",{"src":"/litellm-asset-prefix/_next/static/chunks/3e917c79aadd945b.js","async":true}],["$","script","script-33",{"src":"/litellm-asset-prefix/_next/static/chunks/9bbebdeb3f1cb03f.js","async":true}],"$L6","$L7","$L8","$L9","$La","$Lb","$Lc","$Ld","$Le","$Lf","$L10","$L11","$L12","$L13","$L14","$L15","$L16","$L17","$L18"],"$L19"]}],"loading":null,"isPartial":false}
|
||||
0:{"buildId":"vipo1KaFppvC6fyoT1UMK","rsc":["$","$1","c",{"children":[["$","$L2",null,{"Component":"$3","serverProvidedParams":{"searchParams":{},"params":{},"promises":["$@4","$@5"]}}],[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/3f3fa56b5786d58c.css","precedence":"next"}],["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/0493aafc4891dd29.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/f7e1d08418645368.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/b3d198d6c56a21b8.js","async":true}],["$","script","script-3",{"src":"/litellm-asset-prefix/_next/static/chunks/403c4d96324c23a6.js","async":true}],["$","script","script-4",{"src":"/litellm-asset-prefix/_next/static/chunks/1d1c8edf97a801b6.js","async":true}],["$","script","script-5",{"src":"/litellm-asset-prefix/_next/static/chunks/adb8beb738574863.js","async":true}],["$","script","script-6",{"src":"/litellm-asset-prefix/_next/static/chunks/0549bc9afa7d4888.js","async":true}],["$","script","script-7",{"src":"/litellm-asset-prefix/_next/static/chunks/0b470ffc60999bf4.js","async":true}],["$","script","script-8",{"src":"/litellm-asset-prefix/_next/static/chunks/c847ecdf8c790b0b.js","async":true}],["$","script","script-9",{"src":"/litellm-asset-prefix/_next/static/chunks/e099566e8bd4ee4e.js","async":true}],["$","script","script-10",{"src":"/litellm-asset-prefix/_next/static/chunks/ee5f9a39a526e423.js","async":true}],["$","script","script-11",{"src":"/litellm-asset-prefix/_next/static/chunks/b1c98cc932a0ab19.js","async":true}],["$","script","script-12",{"src":"/litellm-asset-prefix/_next/static/chunks/4e17b625d75327a7.js","async":true}],["$","script","script-13",{"src":"/litellm-asset-prefix/_next/static/chunks/7b788dd93ad868b3.js","async":true}],["$","script","script-14",{"src":"/litellm-asset-prefix/_next/static/chunks/a06cc76a774dd182.js","async":true}],["$","script","script-15",{"src":"/litellm-asset-prefix/_next/static/chunks/ca5fbafaf3826374.js","async":true}],["$","script","script-16",{"src":"/litellm-asset-prefix/_next/static/chunks/7caea73b77a79d3c.js","async":true}],["$","script","script-17",{"src":"/litellm-asset-prefix/_next/static/chunks/10dc4591ef08a91f.js","async":true}],["$","script","script-18",{"src":"/litellm-asset-prefix/_next/static/chunks/0b3d09ff6c6e4335.js","async":true}],["$","script","script-19",{"src":"/litellm-asset-prefix/_next/static/chunks/baadbd26839e7b66.js","async":true}],["$","script","script-20",{"src":"/litellm-asset-prefix/_next/static/chunks/2971c4658f1bcd7d.js","async":true}],["$","script","script-21",{"src":"/litellm-asset-prefix/_next/static/chunks/134f728fa7099e3e.js","async":true}],["$","script","script-22",{"src":"/litellm-asset-prefix/_next/static/chunks/679dbd657c8b5aef.js","async":true}],["$","script","script-23",{"src":"/litellm-asset-prefix/_next/static/chunks/94f7208f5087e27c.js","async":true}],["$","script","script-24",{"src":"/litellm-asset-prefix/_next/static/chunks/43f6fc3c2ab9cf23.js","async":true}],["$","script","script-25",{"src":"/litellm-asset-prefix/_next/static/chunks/4e06277331e725da.js","async":true}],["$","script","script-26",{"src":"/litellm-asset-prefix/_next/static/chunks/3b30ab8eaa03bc21.js","async":true}],["$","script","script-27",{"src":"/litellm-asset-prefix/_next/static/chunks/7e417dd24c8becd0.js","async":true}],["$","script","script-28",{"src":"/litellm-asset-prefix/_next/static/chunks/da1c7742cc6fe8b4.js","async":true}],["$","script","script-29",{"src":"/litellm-asset-prefix/_next/static/chunks/ca7a3fdb635fb7dc.js","async":true}],["$","script","script-30",{"src":"/litellm-asset-prefix/_next/static/chunks/a09028cd611c08ef.js","async":true}],["$","script","script-31",{"src":"/litellm-asset-prefix/_next/static/chunks/77e1b16e6f85230c.js","async":true}],["$","script","script-32",{"src":"/litellm-asset-prefix/_next/static/chunks/ad02f56c287539eb.js","async":true}],["$","script","script-33",{"src":"/litellm-asset-prefix/_next/static/chunks/908828a91f602d8b.js","async":true}],"$L6","$L7","$L8","$L9","$La","$Lb","$Lc","$Ld","$Le","$Lf","$L10","$L11","$L12","$L13","$L14","$L15","$L16","$L17","$L18"],"$L19"]}],"loading":null,"isPartial":false}
|
||||
4:{}
|
||||
5:"$0:rsc:props:children:0:props:serverProvidedParams:params"
|
||||
6:["$","script","script-34",{"src":"/litellm-asset-prefix/_next/static/chunks/0a65da2cd24e2ab6.js","async":true}]
|
||||
7:["$","script","script-35",{"src":"/litellm-asset-prefix/_next/static/chunks/908828a91f602d8b.js","async":true}]
|
||||
8:["$","script","script-36",{"src":"/litellm-asset-prefix/_next/static/chunks/5f2d62a75803a3f7.js","async":true}]
|
||||
9:["$","script","script-37",{"src":"/litellm-asset-prefix/_next/static/chunks/ca5fbafaf3826374.js","async":true}]
|
||||
7:["$","script","script-35",{"src":"/litellm-asset-prefix/_next/static/chunks/fcdf7322b0aa3e2e.js","async":true}]
|
||||
8:["$","script","script-36",{"src":"/litellm-asset-prefix/_next/static/chunks/a8f7c8c5eeb6e042.js","async":true}]
|
||||
9:["$","script","script-37",{"src":"/litellm-asset-prefix/_next/static/chunks/4980372eaa37b78b.js","async":true}]
|
||||
a:["$","script","script-38",{"src":"/litellm-asset-prefix/_next/static/chunks/d3ac82723ec9e30d.js","async":true}]
|
||||
b:["$","script","script-39",{"src":"/litellm-asset-prefix/_next/static/chunks/9b0ee76cbdef1a2a.js","async":true}]
|
||||
b:["$","script","script-39",{"src":"/litellm-asset-prefix/_next/static/chunks/496b84010c33cf69.js","async":true}]
|
||||
c:["$","script","script-40",{"src":"/litellm-asset-prefix/_next/static/chunks/1bc2898be56acd1b.js","async":true}]
|
||||
d:["$","script","script-41",{"src":"/litellm-asset-prefix/_next/static/chunks/fcdf7322b0aa3e2e.js","async":true}]
|
||||
d:["$","script","script-41",{"src":"/litellm-asset-prefix/_next/static/chunks/6188170a32c9a3c3.js","async":true}]
|
||||
e:["$","script","script-42",{"src":"/litellm-asset-prefix/_next/static/chunks/878832edb30e99a4.js","async":true}]
|
||||
f:["$","script","script-43",{"src":"/litellm-asset-prefix/_next/static/chunks/496b84010c33cf69.js","async":true}]
|
||||
10:["$","script","script-44",{"src":"/litellm-asset-prefix/_next/static/chunks/8e3d0ce9505a304f.js","async":true}]
|
||||
f:["$","script","script-43",{"src":"/litellm-asset-prefix/_next/static/chunks/934dbc43f8c1abde.js","async":true}]
|
||||
10:["$","script","script-44",{"src":"/litellm-asset-prefix/_next/static/chunks/a0f7bfbaffe81a17.js","async":true}]
|
||||
11:["$","script","script-45",{"src":"/litellm-asset-prefix/_next/static/chunks/e1f23fd814ac3500.js","async":true}]
|
||||
12:["$","script","script-46",{"src":"/litellm-asset-prefix/_next/static/chunks/88c74f8b4b20d25a.js","async":true}]
|
||||
13:["$","script","script-47",{"src":"/litellm-asset-prefix/_next/static/chunks/99cf9cf99df5ccfc.js","async":true}]
|
||||
14:["$","script","script-48",{"src":"/litellm-asset-prefix/_next/static/chunks/4980372eaa37b78b.js","async":true}]
|
||||
15:["$","script","script-49",{"src":"/litellm-asset-prefix/_next/static/chunks/0cdfadbcf4b8c9e4.js","async":true}]
|
||||
16:["$","script","script-50",{"src":"/litellm-asset-prefix/_next/static/chunks/8f3bf592254c6c3b.js","async":true}]
|
||||
17:["$","script","script-51",{"src":"/litellm-asset-prefix/_next/static/chunks/8c17e934bd227606.js","async":true}]
|
||||
18:["$","script","script-52",{"src":"/litellm-asset-prefix/_next/static/chunks/b98447395b5d37ef.js","async":true}]
|
||||
13:["$","script","script-47",{"src":"/litellm-asset-prefix/_next/static/chunks/b948aa17e97c458d.js","async":true}]
|
||||
14:["$","script","script-48",{"src":"/litellm-asset-prefix/_next/static/chunks/20acf4fa815c638e.js","async":true}]
|
||||
15:["$","script","script-49",{"src":"/litellm-asset-prefix/_next/static/chunks/659ce28f2cb74401.js","async":true}]
|
||||
16:["$","script","script-50",{"src":"/litellm-asset-prefix/_next/static/chunks/d6ab357d1bbb53f0.js","async":true}]
|
||||
17:["$","script","script-51",{"src":"/litellm-asset-prefix/_next/static/chunks/99cf9cf99df5ccfc.js","async":true}]
|
||||
18:["$","script","script-52",{"src":"/litellm-asset-prefix/_next/static/chunks/b3d631e60d6e8e9b.js","async":true}]
|
||||
19:["$","$L1a",null,{"children":["$","$1b",null,{"name":"Next.MetadataOutlet","children":"$@1c"}]}]
|
||||
1c:null
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -3,4 +3,4 @@
|
|||
3:I[897367,["/litellm-asset-prefix/_next/static/chunks/d96012bcfc98706a.js","/litellm-asset-prefix/_next/static/chunks/dbca964212122d58.js"],"MetadataBoundary"]
|
||||
4:"$Sreact.suspense"
|
||||
5:I[27201,["/litellm-asset-prefix/_next/static/chunks/d96012bcfc98706a.js","/litellm-asset-prefix/_next/static/chunks/dbca964212122d58.js"],"IconMark"]
|
||||
0:{"buildId":"8TZ2JbOi7SZ6BCj9ScTHW","rsc":["$","$1","h",{"children":[null,["$","$L2",null,{"children":[["$","meta","0",{"charSet":"utf-8"}],["$","meta","1",{"name":"viewport","content":"width=device-width, initial-scale=1"}]]}],["$","div",null,{"hidden":true,"children":["$","$L3",null,{"children":["$","$4",null,{"name":"Next.Metadata","children":[["$","title","0",{"children":"LiteLLM Dashboard"}],["$","meta","1",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","2",{"rel":"icon","href":"/favicon.ico?favicon.1d32c690.ico","sizes":"48x48","type":"image/x-icon"}],["$","link","3",{"rel":"icon","href":"./favicon.ico"}],["$","$L5","4",{}]]}]}]}],["$","meta",null,{"name":"next-size-adjust","content":""}]]}],"loading":null,"isPartial":false}
|
||||
0:{"buildId":"vipo1KaFppvC6fyoT1UMK","rsc":["$","$1","h",{"children":[null,["$","$L2",null,{"children":[["$","meta","0",{"charSet":"utf-8"}],["$","meta","1",{"name":"viewport","content":"width=device-width, initial-scale=1"}]]}],["$","div",null,{"hidden":true,"children":["$","$L3",null,{"children":["$","$4",null,{"name":"Next.Metadata","children":[["$","title","0",{"children":"LiteLLM Dashboard"}],["$","meta","1",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","2",{"rel":"icon","href":"/favicon.ico?favicon.1d32c690.ico","sizes":"48x48","type":"image/x-icon"}],["$","link","3",{"rel":"icon","href":"./favicon.ico"}],["$","$L5","4",{}]]}]}]}],["$","meta",null,{"name":"next-size-adjust","content":""}]]}],"loading":null,"isPartial":false}
|
||||
|
|
|
|||
|
|
@ -5,4 +5,4 @@
|
|||
5:I[837457,["/litellm-asset-prefix/_next/static/chunks/d96012bcfc98706a.js","/litellm-asset-prefix/_next/static/chunks/dbca964212122d58.js"],"default"]
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/4e20891f2fd03463.css","style"]
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/91037395c95e366d.css","style"]
|
||||
0:{"buildId":"8TZ2JbOi7SZ6BCj9ScTHW","rsc":["$","$1","c",{"children":[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/4e20891f2fd03463.css","precedence":"next"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/91037395c95e366d.css","precedence":"next"}],["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/9e09de50158b3159.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/7e5fe5584502da06.js","async":true}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"inter_5972bc34-module__OU16Qa__className","children":["$","$L2",null,{"children":["$","$L3",null,{"children":["$","$L4",null,{"parallelRouterKey":"children","template":["$","$L5",null,{}],"notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]]}]}]}]}]}]]}],"loading":null,"isPartial":false}
|
||||
0:{"buildId":"vipo1KaFppvC6fyoT1UMK","rsc":["$","$1","c",{"children":[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/4e20891f2fd03463.css","precedence":"next"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/91037395c95e366d.css","precedence":"next"}],["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/9e09de50158b3159.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/7e5fe5584502da06.js","async":true}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"inter_5972bc34-module__OU16Qa__className","children":["$","$L2",null,{"children":["$","$L3",null,{"children":["$","$L4",null,{"parallelRouterKey":"children","template":["$","$L5",null,{}],"notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]]}]}]}]}]}]]}],"loading":null,"isPartial":false}
|
||||
|
|
|
|||
|
|
@ -2,4 +2,4 @@
|
|||
:HL["/litellm-asset-prefix/_next/static/chunks/91037395c95e366d.css","style"]
|
||||
:HL["/litellm-asset-prefix/_next/static/media/83afe278b6a6bb3c-s.p.3a6ba036.woff2","font",{"crossOrigin":"","type":"font/woff2"}]
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/3f3fa56b5786d58c.css","style"]
|
||||
0:{"buildId":"8TZ2JbOi7SZ6BCj9ScTHW","tree":{"name":"","paramType":null,"paramKey":"","hasRuntimePrefetch":false,"slots":{"children":{"name":"__PAGE__","paramType":null,"paramKey":"__PAGE__","hasRuntimePrefetch":false,"slots":null,"isRootLayout":false}},"isRootLayout":true},"staleTime":300}
|
||||
0:{"buildId":"vipo1KaFppvC6fyoT1UMK","tree":{"name":"","paramType":null,"paramKey":"","hasRuntimePrefetch":false,"slots":{"children":{"name":"__PAGE__","paramType":null,"paramKey":"__PAGE__","hasRuntimePrefetch":false,"slots":null,"isRootLayout":false}},"isRootLayout":true},"staleTime":300}
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue