mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Merge pull request #32156 from BerriAI/litellm_internal_staging
Some checks failed
GitHub Actions Security Analysis / zizmor (push) Has been cancelled
CodeQL / Analyze (actions) (push) Has been cancelled
CodeQL / Analyze (javascript-typescript) (push) Has been cancelled
CodeQL / Analyze (python) (push) Has been cancelled
CodSpeed Benchmarks / benchmarks (push) Has been cancelled
Helm unit test / unit-test (push) Has been cancelled
Scorecard supply-chain security / Scorecard analysis (push) Has been cancelled
Some checks failed
GitHub Actions Security Analysis / zizmor (push) Has been cancelled
CodeQL / Analyze (actions) (push) Has been cancelled
CodeQL / Analyze (javascript-typescript) (push) Has been cancelled
CodeQL / Analyze (python) (push) Has been cancelled
CodSpeed Benchmarks / benchmarks (push) Has been cancelled
Helm unit test / unit-test (push) Has been cancelled
Scorecard supply-chain security / Scorecard analysis (push) Has been cancelled
chore(ci): promote internal staging to main
This commit is contained in:
commit
79a6b8f7f0
908 changed files with 18786 additions and 10670 deletions
|
|
@ -1746,13 +1746,13 @@ jobs:
|
|||
-e LANGFUSE_PROJECT1_SECRET=$LANGFUSE_PROJECT1_SECRET \
|
||||
-e LANGFUSE_PROJECT2_SECRET=$LANGFUSE_PROJECT2_SECRET \
|
||||
-e RECORDER_OPENAI_BASE_URL=http://host.docker.internal:8090/v1 \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/proxy_server_config.yaml:/app/config.yaml \
|
||||
my-app:latest \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -1832,13 +1832,13 @@ jobs:
|
|||
-e LANGFUSE_PROJECT2_PUBLIC=$LANGFUSE_PROJECT2_PUBLIC \
|
||||
-e LANGFUSE_PROJECT1_SECRET=$LANGFUSE_PROJECT1_SECRET \
|
||||
-e LANGFUSE_PROJECT2_SECRET=$LANGFUSE_PROJECT2_SECRET \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/oai_misc_config.yaml:/app/config.yaml \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -1911,14 +1911,14 @@ jobs:
|
|||
-e COHERE_API_KEY=$COHERE_API_KEY \
|
||||
-e RECORDER_COHERE_BASE_URL=http://host.docker.internal:8090/__recorder_upstream/api.cohere.com \
|
||||
-e GCS_FLUSH_INTERVAL="1" \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/otel_test_config.yaml:/app/config.yaml \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/custom_guardrail.py:/app/custom_guardrail.py \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -1960,13 +1960,13 @@ jobs:
|
|||
-e OPENAI_API_KEY=$OPENAI_API_KEY \
|
||||
-e FAKE_OPENAI_API_BASE=http://host.docker.internal:8190 \
|
||||
-e LITELLM_LICENSE="bad-license" \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app-3 \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/enterprise_config.yaml:/app/config.yaml \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug
|
||||
--port 4000
|
||||
|
||||
- run:
|
||||
name: Start outputting logs for second container
|
||||
|
|
@ -2041,13 +2041,13 @@ jobs:
|
|||
-e DD_SITE=$DD_SITE \
|
||||
-e AWS_REGION_NAME=$AWS_REGION_NAME \
|
||||
-e PROXY_BATCH_WRITE_AT=2 \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/spend_tracking_config.yaml:/app/config.yaml \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -2117,13 +2117,13 @@ jobs:
|
|||
-e USE_DDTRACE=True \
|
||||
-e DD_API_KEY=$DD_API_KEY \
|
||||
-e DD_SITE=$DD_SITE \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml:/app/config.yaml \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Run Docker container 2
|
||||
command: |
|
||||
|
|
@ -2139,13 +2139,13 @@ jobs:
|
|||
-e USE_DDTRACE=True \
|
||||
-e DD_API_KEY=$DD_API_KEY \
|
||||
-e DD_SITE=$DD_SITE \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app-2 \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml:/app/config.yaml \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4001 \
|
||||
--detailed_debug
|
||||
--port 4001
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -2207,13 +2207,13 @@ jobs:
|
|||
-e LITELLM_MASTER_KEY="sk-1234" \
|
||||
-e FAKE_OPENAI_API_BASE=http://host.docker.internal:8190 \
|
||||
-e LITELLM_LICENSE=$LITELLM_LICENSE \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/store_model_db_config.yaml:/app/config.yaml \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -2289,13 +2289,13 @@ jobs:
|
|||
-e DD_API_KEY=$DD_API_KEY \
|
||||
-e DD_SITE=$DD_SITE \
|
||||
-e GCS_FLUSH_INTERVAL="1" \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/docker/build_from_pip/litellm_config.yaml:/app/config.yaml \
|
||||
my-app:latest \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -2365,14 +2365,14 @@ jobs:
|
|||
-e DD_SITE=$DD_SITE \
|
||||
-e LITELLM_LICENSE=$LITELLM_LICENSE \
|
||||
-e LITELLM_USE_CHAT_COMPLETIONS_URL_FOR_ANTHROPIC_MESSAGES=true \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/pass_through_config.yaml:/app/config.yaml \
|
||||
-v $(pwd)/litellm/proxy/example_config_yaml/custom_auth_basic.py:/app/custom_auth_basic.py \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug \
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -2499,13 +2499,13 @@ jobs:
|
|||
-e AWS_SECRET_ACCESS_KEY=$AWS_SECRET_ACCESS_KEY \
|
||||
-e AWS_REGION_NAME="us-east-1" \
|
||||
-e LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS="True" \
|
||||
-e LITELLM_LOG=ERROR \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--name my-app \
|
||||
-v $(pwd)/tests/proxy_e2e_anthropic_messages_tests/test_config.yaml:/app/config.yaml \
|
||||
litellm-docker-database:ci \
|
||||
--config /app/config.yaml \
|
||||
--port 4000 \
|
||||
--detailed_debug
|
||||
--port 4000
|
||||
- run:
|
||||
name: Start outputting logs
|
||||
command: docker logs -f my-app
|
||||
|
|
@ -2629,7 +2629,7 @@ jobs:
|
|||
cd ui/litellm-dashboard
|
||||
|
||||
CI=true npm run test -- --run \
|
||||
--pool forks --poolOptions.forks.maxForks=8
|
||||
--pool forks --poolOptions.forks.maxForks=6
|
||||
|
||||
e2e_ui_testing:
|
||||
docker:
|
||||
|
|
|
|||
2
.github/pull_request_template.md
vendored
2
.github/pull_request_template.md
vendored
|
|
@ -4,7 +4,7 @@
|
|||
|
||||
## Linear ticket
|
||||
|
||||
<!-- if you are an internal contributor (e.g., your username is postfixed with -berri or -berriai), add "Resolves " followed by the Linear ticket e.g., "Resolves LIT-1234" to magically link the Linear ticket to the GitHub PR -->
|
||||
<!-- if you are an internal contributor, add "Resolves " followed by the Linear ticket e.g., "Resolves LIT-1234" to link the Linear ticket to the GitHub PR. If you don't have one, leave the section blank rather than guessing -->
|
||||
|
||||
## Pre-Submission checklist
|
||||
|
||||
|
|
|
|||
2
.github/workflows/test-linting.yml
vendored
2
.github/workflows/test-linting.yml
vendored
|
|
@ -63,7 +63,7 @@ jobs:
|
|||
env:
|
||||
BASE_SHA: ${{ github.event.pull_request.base.sha }}
|
||||
run: |
|
||||
git diff --name-only "$BASE_SHA"...HEAD -- 'litellm/**/*.py' | grep -v '^litellm/enterprise/' > "$RUNNER_TEMP/ruff_format_files.txt" || true
|
||||
git diff --name-only --diff-filter=ACMR "$BASE_SHA"...HEAD -- 'litellm/**/*.py' | grep -v '^litellm/enterprise/' > "$RUNNER_TEMP/ruff_format_files.txt" || true
|
||||
if [ ! -s "$RUNNER_TEMP/ruff_format_files.txt" ]; then
|
||||
echo "No changed litellm Python files to check with ruff format."
|
||||
exit 0
|
||||
|
|
|
|||
|
|
@ -21,7 +21,9 @@ End-to-end tests belong in `tests/e2e/` and must follow the harness conventions
|
|||
|
||||
When creating PRs, don't set base to `main`. `litellm_internal_staging` serves that purpose
|
||||
|
||||
Always use @.github/pull_request_template.md as a guide for your PR body
|
||||
When writing a PR body, treat the comments and imperative instructions inside @.github/pull_request_template.md as rules to follow, not just layout
|
||||
|
||||
If you're resolving a linear ticket, in the "## Linear ticket" section of the PR, say "Resolves LIT-1234", replacing "LIT-1234" with the actual ticket id that you're resolving. If you don't have the ticket id, don't make one up or search for it. Just leave the section blank
|
||||
|
||||
Never use `pytest` commands or the like as "Screenshots / Proof of Fix". We prefer curl'ing a live proxy instance running on localhost:4000 (I like to run it with `python litellm/proxy/proxy_cli.py --config litellm/proxy/dev_config.yaml --detailed_debug --reload --use_v2_migration_resolver 2>&1 | tee litellm.log`) and showing both the command run and the output. Also, it should hit real LLM provider APIs, not mocks, and cost real $$$ because that is the most realistic test. The proof of fix should be exactly what the end user / customer would see / do. The run logs in PR #27703 is a prime example of how to do it (not a huge fan of using a python test script that future me and the team will have no visibility into; I prefer just curl commands or a short list of bash commands (e.g., using `for`)). If it's a UI thing, just tell me which URLs to go to (e.g., http://localhost:4000/ui/?page=logs), where to click, what fields to fill out, etc. along with the other commands to run in an ordered list, and I'll do it myself and post the screenshots after you make the PR
|
||||
|
||||
|
|
|
|||
|
|
@ -477,9 +477,12 @@ class BaseEmailLogger(CustomLogger):
|
|||
_id = user_info.token or user_info.user_id or "default_id"
|
||||
_cache_key = f"email_budget_alerts:soft_budget_crossed:{_id}"
|
||||
|
||||
# Check if we've already sent this alert
|
||||
result = await _cache.async_get_cache(key=_cache_key)
|
||||
if result is None:
|
||||
send_count = await _cache.async_increment_cache(
|
||||
key=_cache_key,
|
||||
value=1,
|
||||
ttl=EMAIL_BUDGET_ALERT_TTL,
|
||||
)
|
||||
if send_count is None or send_count <= 1:
|
||||
# Create WebhookEvent for soft budget alert
|
||||
event_message = f"Soft Budget Crossed - Total Soft Budget: ${user_info.soft_budget}"
|
||||
webhook_event = WebhookEvent(
|
||||
|
|
@ -508,18 +511,12 @@ class BaseEmailLogger(CustomLogger):
|
|||
await self.send_team_soft_budget_alert_email(webhook_event)
|
||||
else:
|
||||
await self.send_soft_budget_alert_email(webhook_event)
|
||||
|
||||
# Cache the alert to prevent duplicate sends
|
||||
await _cache.async_set_cache(
|
||||
key=_cache_key,
|
||||
value="SENT",
|
||||
ttl=EMAIL_BUDGET_ALERT_TTL,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.error(
|
||||
f"Error sending soft budget alert email: {e}",
|
||||
exc_info=True,
|
||||
)
|
||||
await self._release_budget_alert_claim(_cache, _cache_key)
|
||||
return
|
||||
|
||||
# For max_budget_alert, check if we've already sent an alert
|
||||
|
|
@ -545,9 +542,12 @@ class BaseEmailLogger(CustomLogger):
|
|||
_id = user_info.token or user_info.user_id or "default_id"
|
||||
_cache_key = f"email_budget_alerts:max_budget_alert:{_id}"
|
||||
|
||||
# Check if we've already sent this alert
|
||||
result = await _cache.async_get_cache(key=_cache_key)
|
||||
if result is None:
|
||||
send_count = await _cache.async_increment_cache(
|
||||
key=_cache_key,
|
||||
value=1,
|
||||
ttl=EMAIL_BUDGET_ALERT_TTL,
|
||||
)
|
||||
if send_count is None or send_count <= 1:
|
||||
# Calculate percentage
|
||||
percentage = int(
|
||||
EMAIL_BUDGET_ALERT_MAX_SPEND_ALERT_PERCENTAGE * 100
|
||||
|
|
@ -576,18 +576,12 @@ class BaseEmailLogger(CustomLogger):
|
|||
|
||||
try:
|
||||
await self.send_max_budget_alert_email(webhook_event)
|
||||
|
||||
# Cache the alert to prevent duplicate sends
|
||||
await _cache.async_set_cache(
|
||||
key=_cache_key,
|
||||
value="SENT",
|
||||
ttl=EMAIL_BUDGET_ALERT_TTL,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.error(
|
||||
f"Error sending max budget alert email: {e}",
|
||||
exc_info=True,
|
||||
)
|
||||
await self._release_budget_alert_claim(_cache, _cache_key)
|
||||
return
|
||||
|
||||
async def _handle_multi_threshold_max_budget_alert(
|
||||
|
|
@ -617,10 +611,6 @@ class BaseEmailLogger(CustomLogger):
|
|||
f"email_budget_alerts:max_budget_alert:{threshold_pct}:{_id}"
|
||||
)
|
||||
|
||||
result = await _cache.async_get_cache(key=_cache_key)
|
||||
if result is not None:
|
||||
continue
|
||||
|
||||
# Parse emails + auto-include owner
|
||||
emails = _parse_email_list(raw_emails)
|
||||
if user_info.user_email:
|
||||
|
|
@ -634,6 +624,14 @@ class BaseEmailLogger(CustomLogger):
|
|||
continue
|
||||
recipient_emails = list(set(emails))
|
||||
|
||||
send_count = await _cache.async_increment_cache(
|
||||
key=_cache_key,
|
||||
value=1,
|
||||
ttl=EMAIL_BUDGET_ALERT_TTL,
|
||||
)
|
||||
if send_count is not None and send_count > 1:
|
||||
continue
|
||||
|
||||
event_message = f"Max Budget Alert - {threshold_pct}% of Maximum Budget Reached"
|
||||
webhook_event = WebhookEvent(
|
||||
event="max_budget_alert",
|
||||
|
|
@ -660,16 +658,21 @@ class BaseEmailLogger(CustomLogger):
|
|||
threshold_pct=threshold_pct,
|
||||
recipient_emails=recipient_emails,
|
||||
)
|
||||
await _cache.async_set_cache(
|
||||
key=_cache_key,
|
||||
value="SENT",
|
||||
ttl=EMAIL_BUDGET_ALERT_TTL,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.error(
|
||||
f"Error sending multi-threshold max budget alert email for {threshold_pct}%: {e}",
|
||||
exc_info=True,
|
||||
)
|
||||
await self._release_budget_alert_claim(_cache, _cache_key)
|
||||
|
||||
async def _release_budget_alert_claim(self, cache: DualCache, cache_key: str) -> None:
|
||||
try:
|
||||
await cache.async_delete_cache(key=cache_key)
|
||||
except Exception:
|
||||
verbose_proxy_logger.debug(
|
||||
"Failed to release budget alert claim for %s; it expires with the TTL",
|
||||
cache_key,
|
||||
)
|
||||
|
||||
async def _get_email_params(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
[project]
|
||||
name = "litellm-enterprise"
|
||||
version = "0.1.46"
|
||||
version = "0.1.47"
|
||||
description = "Package for LiteLLM Enterprise features"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.9"
|
||||
|
|
@ -26,7 +26,7 @@ required-version = ">=0.10.9"
|
|||
module-root = ""
|
||||
|
||||
[tool.commitizen]
|
||||
version = "0.1.46"
|
||||
version = "0.1.47"
|
||||
version_files = [
|
||||
"pyproject.toml:^version",
|
||||
"../pyproject.toml:litellm-enterprise==",
|
||||
|
|
|
|||
|
|
@ -1500,6 +1500,7 @@ def completion_cost(
|
|||
custom_llm_provider=custom_llm_provider,
|
||||
litellm_model_name=model,
|
||||
data_residency=data_residency,
|
||||
litellm_logging_obj=litellm_logging_obj,
|
||||
)
|
||||
elif call_type == _MCP_CALL_TYPE:
|
||||
from litellm.proxy._experimental.mcp_server.cost_calculator import (
|
||||
|
|
@ -2302,6 +2303,7 @@ def handle_realtime_stream_cost_calculation(
|
|||
custom_llm_provider: str,
|
||||
litellm_model_name: str,
|
||||
data_residency: Optional[str] = None,
|
||||
litellm_logging_obj: Optional[LitellmLoggingObject] = None,
|
||||
) -> float:
|
||||
"""
|
||||
Handles the cost calculation for realtime stream responses.
|
||||
|
|
@ -2337,14 +2339,25 @@ def handle_realtime_stream_cost_calculation(
|
|||
input_cost_per_token += _input_cost_per_token
|
||||
output_cost_per_token += _output_cost_per_token
|
||||
break # exit if we find a valid model
|
||||
total_cost = input_cost_per_token + output_cost_per_token
|
||||
|
||||
if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results):
|
||||
total_cost += handle_realtime_transcription_cost_calculation(
|
||||
transcription_cost = (
|
||||
handle_realtime_transcription_cost_calculation(
|
||||
results=results,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
litellm_model_name=litellm_model_name,
|
||||
)
|
||||
if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results)
|
||||
else 0.0
|
||||
)
|
||||
total_cost = input_cost_per_token + output_cost_per_token + transcription_cost
|
||||
|
||||
_store_cost_breakdown_in_logging_obj(
|
||||
litellm_logging_obj=litellm_logging_obj,
|
||||
prompt_tokens_cost_usd_dollar=input_cost_per_token,
|
||||
completion_tokens_cost_usd_dollar=output_cost_per_token,
|
||||
cost_for_built_in_tools_cost_usd_dollar=0.0,
|
||||
total_cost_usd_dollar=total_cost,
|
||||
additional_costs={"transcription_cost": transcription_cost} if transcription_cost > 0 else None,
|
||||
)
|
||||
|
||||
return total_cost
|
||||
|
||||
|
|
|
|||
|
|
@ -520,13 +520,28 @@ class MCPClient:
|
|||
# Return empty list instead of raising to allow graceful degradation
|
||||
return []
|
||||
|
||||
@staticmethod
|
||||
def error_tool_result(exc: Exception) -> MCPCallToolResult:
|
||||
"""The error result ``call_tool`` returns when it swallows a failure (no re-execution)."""
|
||||
return MCPCallToolResult(
|
||||
content=[TextContent(type="text", text=f"{type(exc).__name__}: {str(exc)}")],
|
||||
isError=True,
|
||||
)
|
||||
|
||||
async def call_tool(
|
||||
self,
|
||||
call_tool_request_params: MCPCallToolRequestParams,
|
||||
host_progress_callback: Optional[Callable] = None,
|
||||
raise_on_error: bool = False,
|
||||
) -> MCPCallToolResult:
|
||||
"""
|
||||
Call an MCP Tool.
|
||||
|
||||
Args:
|
||||
raise_on_error: When True, re-raise the underlying exception instead of returning an
|
||||
``isError=True`` result. The token-exchange (OBO) tool-call path uses this to detect
|
||||
an upstream 401 so it can re-mint the exchanged token and retry once; every other
|
||||
caller keeps the default and gets graceful ``isError`` degradation.
|
||||
"""
|
||||
verbose_logger.info(f"MCP client calling tool '{call_tool_request_params.name}'")
|
||||
|
||||
|
|
@ -579,11 +594,10 @@ class MCPClient:
|
|||
"MCP client detected broken connection/stream - "
|
||||
"the MCP server may have crashed, disconnected, or timed out."
|
||||
)
|
||||
if raise_on_error:
|
||||
raise
|
||||
# Return a default error result instead of raising
|
||||
return MCPCallToolResult(
|
||||
content=[TextContent(type="text", text=f"{error_type}: {str(e)}")], # Empty content for error case
|
||||
isError=True,
|
||||
)
|
||||
return self.error_tool_result(e)
|
||||
|
||||
async def list_prompts(self) -> List[Prompt]:
|
||||
"""List available prompts from the server."""
|
||||
|
|
|
|||
|
|
@ -2043,6 +2043,43 @@ class PrometheusLogger(CustomLogger):
|
|||
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def _extract_api_provider_from_request_data(request_data: dict) -> Optional[str]:
|
||||
"""
|
||||
Best-effort provider for the client-side failure path.
|
||||
|
||||
A request can fail before a deployment is resolved, so the provider is
|
||||
not always known. Prefer the resolved ``custom_llm_provider`` on
|
||||
``litellm_params``, then any provider recovered onto a partial
|
||||
``standard_logging_object`` (e.g. a stream that broke mid-flight), and
|
||||
finally infer it from the requested model name (e.g. ``gpt-4o-mini`` ->
|
||||
``openai``) since the proxy's failure ``request_data`` usually carries
|
||||
only the client-supplied model. Return ``None`` when it cannot be
|
||||
determined so the label emits empty rather than a guess.
|
||||
"""
|
||||
litellm_params = request_data.get("litellm_params") or {}
|
||||
provider = litellm_params.get("custom_llm_provider")
|
||||
if provider:
|
||||
return provider
|
||||
standard_logging_object = request_data.get("standard_logging_object") or {}
|
||||
provider = standard_logging_object.get("custom_llm_provider")
|
||||
if provider:
|
||||
return provider
|
||||
model = litellm_params.get("model") or request_data.get("model")
|
||||
if not model:
|
||||
return None
|
||||
try:
|
||||
return litellm.get_llm_provider(model=model)[1] or None
|
||||
except litellm.exceptions.BadRequestError:
|
||||
return None
|
||||
except Exception as e: # noqa: BLE001 - metrics labeling must never break request/failure handling
|
||||
verbose_logger.debug(
|
||||
"prometheus: unexpected error inferring api_provider from model=%s: %s",
|
||||
model,
|
||||
e,
|
||||
)
|
||||
return None
|
||||
|
||||
async def async_post_call_failure_hook(
|
||||
self,
|
||||
request_data: dict,
|
||||
|
|
@ -2078,6 +2115,7 @@ class PrometheusLogger(CustomLogger):
|
|||
_metadata = request_data.get("metadata", {}) or {}
|
||||
model_id = _metadata.get("model_info", {}).get("id") or request_data.get("model_info", {}).get("id")
|
||||
rate_limit_category, rate_limit_type = self._extract_rate_limit_labels(original_exception)
|
||||
api_provider = self._extract_api_provider_from_request_data(request_data)
|
||||
enum_values = UserAPIKeyLabelValues(
|
||||
end_user=user_api_key_dict.end_user_id,
|
||||
user=user_api_key_dict.user_id,
|
||||
|
|
@ -2099,6 +2137,7 @@ class PrometheusLogger(CustomLogger):
|
|||
client_ip=_metadata.get("requester_ip_address"),
|
||||
user_agent=_metadata.get("user_agent"),
|
||||
model_id=model_id,
|
||||
api_provider=api_provider,
|
||||
stream=(str(request_data.get("stream")) if litellm.prometheus_emit_stream_label else None),
|
||||
)
|
||||
_label_ctx = PrometheusLabelFactoryContext(enum_values)
|
||||
|
|
|
|||
|
|
@ -131,8 +131,26 @@ class SensitiveDataMasker:
|
|||
|
||||
return masked_data
|
||||
|
||||
def mask(self, data: object) -> object:
|
||||
if isinstance(data, Mapping):
|
||||
return self.mask_dict(dict(data))
|
||||
if isinstance(data, list):
|
||||
return self._mask_sequence(
|
||||
data,
|
||||
0,
|
||||
DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER,
|
||||
None,
|
||||
False,
|
||||
)
|
||||
return data
|
||||
|
||||
|
||||
_default_masker = SensitiveDataMasker()
|
||||
_error_masker = SensitiveDataMasker(visible_prefix=4, visible_suffix=0)
|
||||
|
||||
|
||||
def mask_sensitive_structure(data: object) -> object:
|
||||
return _error_masker.mask(data)
|
||||
|
||||
|
||||
def mask_sensitive_keys(data: Dict[str, Any], sensitive_fields: Set[str]) -> Dict[str, Any]:
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ from litellm.types.llms.openai import (
|
|||
ChatCompletionAudioDelta,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
CacheCreationTokenDetails,
|
||||
ChatCompletionAudioResponse,
|
||||
ChatCompletionMessageToolCall,
|
||||
Choices,
|
||||
|
|
@ -541,6 +542,12 @@ class ChunkProcessor:
|
|||
web_search_requests: Optional[int] = None
|
||||
completion_tokens_details: Optional[CompletionTokensDetails] = None
|
||||
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None
|
||||
# Anthropic emits the cache-creation TTL breakdown (5m/1h split) only on
|
||||
# the `message_start` event; the later `message_delta` carries the flat
|
||||
# cache-creation count but drops the nested breakdown. prompt_tokens_details
|
||||
# is last-wins, so without preserving this separately the 1h breakdown is
|
||||
# lost and 1h cache writes get billed at the 5m rate.
|
||||
cache_creation_token_details: Optional[CacheCreationTokenDetails] = None
|
||||
for chunk in chunks:
|
||||
usage_chunk: Optional[Usage] = None
|
||||
if "usage" in chunk:
|
||||
|
|
@ -594,7 +601,18 @@ class ChunkProcessor:
|
|||
"web_search_requests",
|
||||
)
|
||||
|
||||
prompt_tokens_details = usage_chunk_dict["prompt_tokens_details"]
|
||||
prompt_tokens_details = cast(
|
||||
Optional[PromptTokensDetailsWrapper],
|
||||
usage_chunk_dict["prompt_tokens_details"],
|
||||
)
|
||||
|
||||
cache_creation_token_details = self._capture_cache_creation_token_details(
|
||||
prompt_tokens_details, cache_creation_token_details
|
||||
)
|
||||
|
||||
prompt_tokens_details = self._attach_cache_creation_token_details(
|
||||
prompt_tokens_details, cache_creation_token_details
|
||||
)
|
||||
|
||||
completion_tokens = self._reset_anthropic_cursor_completion_tokens(
|
||||
chunks=chunks,
|
||||
|
|
@ -613,6 +631,34 @@ class ChunkProcessor:
|
|||
prompt_tokens_details=prompt_tokens_details,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _capture_cache_creation_token_details(
|
||||
prompt_tokens_details: Optional[PromptTokensDetailsWrapper],
|
||||
current: Optional[CacheCreationTokenDetails],
|
||||
) -> Optional[CacheCreationTokenDetails]:
|
||||
incoming = cast(
|
||||
Optional[CacheCreationTokenDetails],
|
||||
getattr(prompt_tokens_details, "cache_creation_token_details", None),
|
||||
)
|
||||
if incoming is not None:
|
||||
return incoming
|
||||
return current
|
||||
|
||||
@staticmethod
|
||||
def _attach_cache_creation_token_details(
|
||||
prompt_tokens_details: Optional[PromptTokensDetailsWrapper],
|
||||
cache_creation_token_details: Optional[CacheCreationTokenDetails],
|
||||
) -> Optional[PromptTokensDetailsWrapper]:
|
||||
if prompt_tokens_details is None or cache_creation_token_details is None:
|
||||
return prompt_tokens_details
|
||||
existing = cast(
|
||||
Optional[CacheCreationTokenDetails],
|
||||
getattr(prompt_tokens_details, "cache_creation_token_details", None),
|
||||
)
|
||||
if existing is not None:
|
||||
return prompt_tokens_details
|
||||
return prompt_tokens_details.model_copy(update={"cache_creation_token_details": cache_creation_token_details})
|
||||
|
||||
@staticmethod
|
||||
def _reset_anthropic_cursor_completion_tokens(
|
||||
chunks: list[dict[str, Any] | ModelResponse],
|
||||
|
|
|
|||
|
|
@ -78,7 +78,7 @@ async def _prepare_context_managed_request(
|
|||
system: Optional[Any],
|
||||
context_management_spec: Any,
|
||||
litellm_metadata: Optional[Dict],
|
||||
drop_params: Optional[bool],
|
||||
additional_drop_params: Optional[list[str]],
|
||||
llm_router: Any,
|
||||
user_api_key_auth: Any = None,
|
||||
) -> Optional[PolyfillResult]:
|
||||
|
|
@ -95,7 +95,7 @@ async def _prepare_context_managed_request(
|
|||
# silently drop intermediate turns.
|
||||
polyfill_will_run = _polyfill_will_run(
|
||||
context_management_spec=context_management_spec,
|
||||
drop_params=drop_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
)
|
||||
|
||||
if polyfill_will_run:
|
||||
|
|
@ -117,7 +117,7 @@ async def _prepare_context_managed_request(
|
|||
system=working_system,
|
||||
context_management_spec=context_management_spec,
|
||||
litellm_metadata=litellm_metadata,
|
||||
drop_params=drop_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
llm_router=llm_router,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
|
|
@ -143,18 +143,19 @@ async def _prepare_context_managed_request(
|
|||
def _polyfill_will_run(
|
||||
*,
|
||||
context_management_spec: Any,
|
||||
drop_params: Optional[bool],
|
||||
additional_drop_params: Optional[list[str]],
|
||||
) -> bool:
|
||||
"""Return True when ``compact_20260112`` will run via the polyfill dispatcher.
|
||||
|
||||
Mirrors the gating in ``_run_polyfill_if_enabled``: an empty spec or
|
||||
effective ``drop_params`` short-circuits the polyfill. The pre-processing
|
||||
skip only applies when the dispatcher will actually invoke
|
||||
``apply_compact_20260112`` (which has its own compaction-block slicing).
|
||||
Mirrors the gating in ``_run_polyfill_if_enabled``: an empty spec or an
|
||||
explicit ``context_management`` entry in ``additional_drop_params``
|
||||
short-circuits the polyfill. The pre-processing skip only applies when the
|
||||
dispatcher will actually invoke ``apply_compact_20260112`` (which has its
|
||||
own compaction-block slicing).
|
||||
"""
|
||||
edits = _normalize_spec_edits(
|
||||
context_management_spec=context_management_spec,
|
||||
drop_params=drop_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
)
|
||||
if edits is None:
|
||||
return False
|
||||
|
|
@ -169,7 +170,7 @@ def _polyfill_will_run(
|
|||
def _spec_has_non_compact_edits(
|
||||
*,
|
||||
context_management_spec: Any,
|
||||
drop_params: Optional[bool],
|
||||
additional_drop_params: Optional[list[str]],
|
||||
) -> bool:
|
||||
"""Return True when the spec includes edits other than ``compact_20260112``.
|
||||
|
||||
|
|
@ -180,7 +181,7 @@ def _spec_has_non_compact_edits(
|
|||
"""
|
||||
edits = _normalize_spec_edits(
|
||||
context_management_spec=context_management_spec,
|
||||
drop_params=drop_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
)
|
||||
if edits is None:
|
||||
return False
|
||||
|
|
@ -195,10 +196,22 @@ def _spec_has_non_compact_edits(
|
|||
)
|
||||
|
||||
|
||||
def _context_management_explicitly_dropped(additional_drop_params: Optional[list[str]]) -> bool:
|
||||
"""True when the caller opted out of context_management via ``additional_drop_params``.
|
||||
|
||||
``drop_params`` deliberately does NOT gate the polyfill: ``context_management``
|
||||
is a LiteLLM-supported param (native on Anthropic, polyfilled elsewhere), and
|
||||
``drop_params`` only exists to drop genuinely unsupported params.
|
||||
"""
|
||||
if not isinstance(additional_drop_params, list):
|
||||
return False
|
||||
return "context_management" in additional_drop_params
|
||||
|
||||
|
||||
def _normalize_spec_edits(
|
||||
*,
|
||||
context_management_spec: Any,
|
||||
drop_params: Optional[bool],
|
||||
additional_drop_params: Optional[list[str]],
|
||||
) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Return the normalized ``edits`` list, or ``None`` if the polyfill won't run.
|
||||
|
||||
|
|
@ -208,8 +221,7 @@ def _normalize_spec_edits(
|
|||
if not context_management_spec:
|
||||
return None
|
||||
|
||||
effective_drop_params = drop_params if drop_params is not None else litellm.drop_params
|
||||
if effective_drop_params:
|
||||
if _context_management_explicitly_dropped(additional_drop_params):
|
||||
return None
|
||||
|
||||
from litellm.llms.anthropic.experimental_pass_through.context_management.dispatcher import (
|
||||
|
|
@ -230,22 +242,23 @@ async def _run_polyfill_if_enabled(
|
|||
system: Optional[Any],
|
||||
context_management_spec: Any,
|
||||
litellm_metadata: Optional[Dict],
|
||||
drop_params: Optional[bool],
|
||||
additional_drop_params: Optional[list[str]],
|
||||
llm_router: Any,
|
||||
user_api_key_auth: Any = None,
|
||||
) -> Optional[PolyfillResult]:
|
||||
"""Run the async context_management polyfill if a spec is present.
|
||||
|
||||
Returns ``None`` when the spec is empty or drop_params is on. Raises
|
||||
``AnthropicContextManagementError`` so the /v1/messages endpoint can
|
||||
emit an Anthropic-format 400. All other exceptions are best-effort
|
||||
swallowed (matches v0 behavior).
|
||||
Returns ``None`` when the spec is empty or ``context_management`` is
|
||||
listed in ``additional_drop_params`` (the explicit opt-out; ``drop_params``
|
||||
does not disable the polyfill because context_management is a supported
|
||||
param). Raises ``AnthropicContextManagementError`` so the /v1/messages
|
||||
endpoint can emit an Anthropic-format 400. All other exceptions are
|
||||
best-effort swallowed (matches v0 behavior).
|
||||
"""
|
||||
if not context_management_spec:
|
||||
return None
|
||||
|
||||
effective_drop_params = drop_params if drop_params is not None else litellm.drop_params
|
||||
if effective_drop_params:
|
||||
if _context_management_explicitly_dropped(additional_drop_params):
|
||||
return None
|
||||
|
||||
try:
|
||||
|
|
@ -274,7 +287,7 @@ async def _run_polyfill_if_enabled(
|
|||
# emits an Anthropic-format error.
|
||||
if _spec_has_non_compact_edits(
|
||||
context_management_spec=context_management_spec,
|
||||
drop_params=drop_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
):
|
||||
raise AnthropicContextManagementError(
|
||||
status_code=500,
|
||||
|
|
@ -533,7 +546,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
) -> Union[AnthropicMessagesResponse, AsyncIterator[Any], Iterator[bytes]]:
|
||||
"""Handle non-Anthropic models asynchronously using the adapter"""
|
||||
context_management = kwargs.pop("context_management", None)
|
||||
drop_params: Optional[bool] = kwargs.get("drop_params", None)
|
||||
additional_drop_params: Optional[list[str]] = kwargs.get("additional_drop_params", None)
|
||||
litellm_router = kwargs.pop("litellm_router", None)
|
||||
if litellm_router is None:
|
||||
try:
|
||||
|
|
@ -555,7 +568,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
system=system,
|
||||
context_management_spec=context_management,
|
||||
litellm_metadata=proxy_litellm_metadata,
|
||||
drop_params=drop_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
llm_router=litellm_router,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
|
|
@ -661,7 +674,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
# ``compact_20260112`` editor can ``await`` the summarization model);
|
||||
# bridge to it via ``run_async_function``.
|
||||
context_management = kwargs.pop("context_management", None)
|
||||
drop_params: Optional[bool] = kwargs.get("drop_params", None)
|
||||
additional_drop_params: Optional[list[str]] = kwargs.get("additional_drop_params", None)
|
||||
# Deliberately do NOT auto-attach the proxy ``llm_router`` here:
|
||||
# ``run_async_function`` spawns a new event loop in a worker thread
|
||||
# to bridge to the async dispatcher, but the proxy router's httpx
|
||||
|
|
@ -696,7 +709,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
system=system,
|
||||
context_management_spec=context_management,
|
||||
litellm_metadata=proxy_litellm_metadata,
|
||||
drop_params=drop_params,
|
||||
additional_drop_params=additional_drop_params,
|
||||
llm_router=litellm_router,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -17,7 +17,9 @@ How it works:
|
|||
import uuid
|
||||
from typing import Any, AsyncIterator, Dict, List, Optional, Union
|
||||
|
||||
import litellm
|
||||
import litellm.constants as _c
|
||||
from litellm.litellm_core_utils.url_utils import validate_url
|
||||
from litellm.llms.anthropic.common_utils import strip_advisor_blocks_from_messages
|
||||
from litellm.types.llms.anthropic_messages.anthropic_response import (
|
||||
AnthropicMessagesResponse,
|
||||
|
|
@ -76,16 +78,7 @@ class AdvisorOrchestrationHandler(MessagesInterceptor):
|
|||
raise ValueError("advisor tool definition must include a 'model' field specifying the advisor model")
|
||||
_raw_max_uses = advisor_tool.get("max_uses")
|
||||
max_uses: int = ADVISOR_MAX_USES if _raw_max_uses is None else int(_raw_max_uses)
|
||||
# Optional routing overrides for the advisor sub-call (e.g. proxy routing).
|
||||
# If not set in the tool definition, litellm resolves from env vars.
|
||||
# The advisor tool is caller-controlled; only honor a client-supplied
|
||||
# api_base/api_key when the proxy has enabled clientside credentials,
|
||||
# otherwise let litellm resolve from server config.
|
||||
advisor_api_key: Optional[str] = None
|
||||
advisor_api_base: Optional[str] = None
|
||||
if _allow_client_side_advisor_credentials():
|
||||
advisor_api_key = advisor_tool.get("api_key")
|
||||
advisor_api_base = advisor_tool.get("api_base")
|
||||
advisor_api_key, advisor_api_base = _resolve_advisor_credentials(advisor_tool)
|
||||
|
||||
# Build the synthetic tool definition the provider will receive.
|
||||
synthetic_advisor_tool = _make_synthetic_advisor_tool()
|
||||
|
|
@ -186,6 +179,49 @@ def _allow_client_side_advisor_credentials() -> bool:
|
|||
return general_settings.get("allow_client_side_credentials") is True
|
||||
|
||||
|
||||
def _resolve_advisor_credentials(advisor_tool: dict) -> tuple[Optional[str], Optional[str]]:
|
||||
"""Resolve the (api_key, api_base) override for the advisor sub-call.
|
||||
|
||||
A caller-supplied ``api_base`` is only honored alongside a caller-supplied
|
||||
``api_key``: without one, ``AnthropicModelInfo.get_auth_header()`` falls
|
||||
back to the proxy's own Anthropic credentials, which would then be sent to
|
||||
the caller-chosen ``api_base``. A caller-supplied ``api_base`` is also
|
||||
required to be https with TLS verification on, and SSRF-validated so it
|
||||
can't target a private/internal/cloud-metadata address, mirroring
|
||||
``proxy.auth.auth_utils.check_complete_credentials``. https with TLS
|
||||
verification is required because ``validate_url`` only rewrites the
|
||||
connection to a DNS-pinned IP for http, or for https with
|
||||
``litellm.ssl_verify`` disabled; otherwise it returns the URL unchanged
|
||||
and relies on certificate validation to block DNS rebinding, so this
|
||||
closes the same gap without threading the pinned URL through the whole
|
||||
``anthropic_messages()`` call chain.
|
||||
"""
|
||||
if not _allow_client_side_advisor_credentials():
|
||||
return None, None
|
||||
api_key: Optional[str] = advisor_tool.get("api_key")
|
||||
api_base: Optional[str] = advisor_tool.get("api_base")
|
||||
if api_base is None:
|
||||
return api_key, None
|
||||
if not api_key:
|
||||
raise ValueError(
|
||||
"advisor tool definition sets 'api_base' without 'api_key'. A "
|
||||
"caller-supplied api_base is only honored alongside a "
|
||||
"caller-supplied api_key, so the proxy's own credentials are "
|
||||
"never sent to a caller-chosen destination."
|
||||
)
|
||||
if not api_base.startswith("https://"):
|
||||
raise ValueError(f"advisor tool definition sets 'api_base'={api_base!r}, which must use the https scheme.")
|
||||
if getattr(litellm, "ssl_verify", True) is False:
|
||||
raise ValueError(
|
||||
"advisor tool definition sets 'api_base' but the proxy has TLS verification "
|
||||
"disabled (litellm.ssl_verify=False), so a caller-supplied api_base can't be "
|
||||
"safely validated against DNS rebinding."
|
||||
)
|
||||
if getattr(litellm, "user_url_validation", True):
|
||||
validate_url(api_base)
|
||||
return api_key, api_base
|
||||
|
||||
|
||||
def _make_synthetic_advisor_tool() -> Dict:
|
||||
"""Build a regular tool definition the executor provider can understand."""
|
||||
return {
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from litellm.proxy._types import (
|
|||
LiteLLM_TeamTable,
|
||||
ProxyException,
|
||||
SpecialHeaders,
|
||||
SpecialMCPServerName,
|
||||
SpecialMCPServerNames,
|
||||
UserAPIKeyAuth,
|
||||
)
|
||||
|
|
@ -1041,6 +1042,9 @@ class MCPRequestHandler:
|
|||
if object_permissions is None:
|
||||
return list(set(team_access_group_servers))
|
||||
|
||||
if SpecialMCPServerName.all_proxy_servers.value in (object_permissions.mcp_servers or []):
|
||||
return list(global_mcp_server_manager.get_registry().keys())
|
||||
|
||||
direct_mcp_servers = global_mcp_server_manager.expand_permission_list(object_permissions.mcp_servers or [])
|
||||
|
||||
legacy_access_group_servers = await MCPRequestHandler._get_mcp_servers_from_access_groups(
|
||||
|
|
|
|||
|
|
@ -1288,7 +1288,14 @@ async def _build_oauth_protected_resource_response(
|
|||
detail=(f"Upstream oauth-protected-resource metadata unavailable for MCP server {mcp_server.name!r}"),
|
||||
)
|
||||
|
||||
_raise_unless_oauth2_discovery_server(mcp_server, mcp_server_name, "not an OAuth-protected resource")
|
||||
obo_response = _obo_protected_resource_response(mcp_server, resource_url)
|
||||
if obo_response is not None:
|
||||
return obo_response
|
||||
|
||||
# An OBO server with no configured issuer falls through to the gateway default so discovery still
|
||||
# returns metadata; every other non-oauth2 named server 404s to avoid enumeration.
|
||||
if mcp_server is None or mcp_server.auth_type != MCPAuth.oauth2_token_exchange:
|
||||
_raise_unless_oauth2_discovery_server(mcp_server, mcp_server_name, "not an OAuth-protected resource")
|
||||
|
||||
return {
|
||||
"authorization_servers": [
|
||||
|
|
@ -1299,6 +1306,51 @@ async def _build_oauth_protected_resource_response(
|
|||
}
|
||||
|
||||
|
||||
def _obo_protected_resource_response(mcp_server: Optional[MCPServer], resource_url: str) -> Optional[dict]:
|
||||
"""The OBO (token_exchange) PRM, or None when this server is not OBO / no issuer is configured.
|
||||
|
||||
The client SSOs with the IdP to obtain a subject token, which LiteLLM then exchanges, so discovery
|
||||
points at the JWT-auth issuer(s) LiteLLM trusts (the same IdP that issues and validates the
|
||||
subject), not the gateway. None falls the caller back to the gateway default so discovery still
|
||||
returns metadata; it just can't name the IdP.
|
||||
"""
|
||||
if mcp_server is None or mcp_server.auth_type != MCPAuth.oauth2_token_exchange:
|
||||
return None
|
||||
issuers = _jwt_auth_issuers()
|
||||
if not issuers:
|
||||
return None
|
||||
return {
|
||||
"authorization_servers": issuers,
|
||||
"resource": resource_url,
|
||||
"scopes_supported": (mcp_server.scopes if mcp_server.scopes else []),
|
||||
}
|
||||
|
||||
|
||||
def _jwt_auth_issuers() -> list:
|
||||
"""The OAuth issuer identifier(s) LiteLLM's JWT auth trusts, for the OBO PRM authorization_servers.
|
||||
|
||||
In token_exchange the IdP that issues the subject JWT is the same one LiteLLM validates it
|
||||
against, so OBO discovery points clients at the JWT-auth issuer to obtain a subject token.
|
||||
Sourced from ``JWT_ISSUER`` and any configured ``litellm_jwtauth.issuers``.
|
||||
"""
|
||||
import os # noqa: PLC0415
|
||||
|
||||
from litellm.proxy.proxy_server import general_settings # noqa: PLC0415
|
||||
|
||||
issuers: list = []
|
||||
env_issuer = os.getenv("JWT_ISSUER")
|
||||
if env_issuer:
|
||||
issuers.append(env_issuer)
|
||||
|
||||
jwtauth = general_settings.get("litellm_jwtauth") if isinstance(general_settings, dict) else None
|
||||
raw_issuers = jwtauth.get("issuers") if isinstance(jwtauth, dict) else getattr(jwtauth, "issuers", None)
|
||||
for cfg in raw_issuers or []:
|
||||
issuer = cfg.get("issuer") if isinstance(cfg, dict) else getattr(cfg, "issuer", None)
|
||||
if issuer and issuer not in issuers:
|
||||
issuers.append(issuer)
|
||||
return issuers
|
||||
|
||||
|
||||
# Standard MCP pattern: /.well-known/oauth-protected-resource/mcp/{server_name}
|
||||
# This is the pattern expected by standard MCP clients (mcp-inspector, VSCode Copilot)
|
||||
@router.get(
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from typing import Any, AsyncIterator, Callable, Literal, Optional, Union, cast
|
|||
from urllib.parse import urlparse
|
||||
|
||||
import anyio
|
||||
import httpx
|
||||
from fastapi import HTTPException
|
||||
from httpx import HTTPStatusError
|
||||
from mcp import ReadResourceResult, Resource
|
||||
|
|
@ -64,6 +65,7 @@ from litellm.proxy._experimental.mcp_server.outbound_credentials import (
|
|||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.adapter import (
|
||||
raise_public,
|
||||
raise_token_exchange_challenge,
|
||||
raise_user_oauth_challenge,
|
||||
to_server_spec,
|
||||
to_subject,
|
||||
|
|
@ -71,8 +73,13 @@ from litellm.proxy._experimental.mcp_server.outbound_credentials.adapter import
|
|||
from litellm.proxy._experimental.mcp_server.outbound_credentials.per_user_oauth_store import (
|
||||
LazyPerUserOAuthTokenStore,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.token_exchange_provider import (
|
||||
build_token_exchanger,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.types import (
|
||||
AuthorizationCodeConfig,
|
||||
ServerSpec,
|
||||
TokenExchangeConfig,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.utils import (
|
||||
MCP_TOOL_PREFIX_SEPARATOR,
|
||||
|
|
@ -105,7 +112,7 @@ from litellm.proxy._types import (
|
|||
from litellm.proxy.auth.ip_address_utils import IPAddressUtils
|
||||
from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_value_helper
|
||||
from litellm.proxy.common_utils.user_api_key_cache import get_management_object_ttl
|
||||
from litellm.proxy.utils import ProxyLogging
|
||||
from litellm.proxy.utils import ProxyLogging, get_server_root_path
|
||||
from litellm.repositories.table_repositories import MCPServerRepository
|
||||
from litellm.types.llms.custom_http import httpxSpecialProvider
|
||||
from litellm.types.mcp import MCPAuth, MCPStdioConfig
|
||||
|
|
@ -207,6 +214,10 @@ def _should_strip_caller_authorization(
|
|||
``Authorization`` is the upstream OAuth token and must be
|
||||
forwarded, so we keep it.
|
||||
"""
|
||||
if mcp_server.auth_type == MCPAuth.oauth2_token_exchange:
|
||||
# OBO: the inbound Authorization is the subject token. It is exchanged at the IdP and only the
|
||||
# exchanged token is sent upstream, so the raw caller bearer must never be forwarded.
|
||||
return True
|
||||
if mcp_server.has_client_credentials:
|
||||
return True
|
||||
if mcp_server.auth_type == MCPAuth.oauth2 and to_server_spec(mcp_server) is not None:
|
||||
|
|
@ -529,9 +540,22 @@ class MCPServerManager:
|
|||
return "client_credentials"
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _obo_needs_endpoint_discovery(
|
||||
auth_type: Optional[MCPAuthType],
|
||||
token_exchange_endpoint: Optional[str],
|
||||
token_url: Optional[str],
|
||||
) -> bool:
|
||||
"""An ``oauth2_token_exchange`` server with no configured token endpoint can have it
|
||||
discovered (RFC 9728 -> RFC 8414) the same way the ``oauth2`` flow already does; an explicitly
|
||||
configured ``token_exchange_endpoint``/``token_url`` wins and skips the discovery round-trip.
|
||||
"""
|
||||
return auth_type == MCPAuth.oauth2_token_exchange and not (token_exchange_endpoint or token_url)
|
||||
|
||||
def __init__(self, cred_provider: Optional[UpstreamCredentialProvider] = None):
|
||||
self._cred_provider = cred_provider or UpstreamCredentialProvider(
|
||||
oauth_token_store=LazyPerUserOAuthTokenStore(self.get_mcp_server_by_id)
|
||||
oauth_token_store=LazyPerUserOAuthTokenStore(self.get_mcp_server_by_id),
|
||||
token_exchanger=build_token_exchanger(),
|
||||
)
|
||||
self.registry: dict[str, MCPServer] = {}
|
||||
self.config_mcp_servers: dict[str, MCPServer] = {}
|
||||
|
|
@ -717,9 +741,17 @@ class MCPServerManager:
|
|||
)
|
||||
|
||||
auth_type = server_config.get("auth_type", None)
|
||||
if server_url and auth_type is not None and auth_type == MCPAuth.oauth2:
|
||||
if server_url and (
|
||||
auth_type == MCPAuth.oauth2
|
||||
or self._obo_needs_endpoint_discovery(
|
||||
auth_type,
|
||||
server_config.get("token_exchange_endpoint"),
|
||||
server_config.get("token_url"),
|
||||
)
|
||||
):
|
||||
mcp_oauth_metadata = await self._descovery_metadata(
|
||||
server_url=server_url,
|
||||
allow_origin_fallback=auth_type == MCPAuth.oauth2,
|
||||
)
|
||||
else:
|
||||
mcp_oauth_metadata = None
|
||||
|
|
@ -1097,9 +1129,19 @@ class MCPServerManager:
|
|||
|
||||
auth_type = cast(MCPAuthType, mcp_server.auth_type)
|
||||
server_url = mcp_server.url
|
||||
needs_discovery = bool(server_url) and auth_type == MCPAuth.oauth2 and not mcp_server.authorization_url
|
||||
needs_discovery = bool(server_url) and (
|
||||
(auth_type == MCPAuth.oauth2 and not mcp_server.authorization_url)
|
||||
or self._obo_needs_endpoint_discovery(
|
||||
auth_type,
|
||||
credentials_dict.get("token_exchange_endpoint") if credentials_dict else None,
|
||||
mcp_server.token_url,
|
||||
)
|
||||
)
|
||||
mcp_oauth_metadata = (
|
||||
await self._descovery_metadata(server_url=server_url) # type: ignore[arg-type]
|
||||
await self._descovery_metadata(
|
||||
server_url=server_url, # type: ignore[arg-type]
|
||||
allow_origin_fallback=auth_type == MCPAuth.oauth2,
|
||||
)
|
||||
if needs_discovery
|
||||
else None
|
||||
)
|
||||
|
|
@ -1174,8 +1216,49 @@ class MCPServerManager:
|
|||
max_concurrent_requests=getattr(mcp_server, "max_concurrent_requests", None),
|
||||
)
|
||||
_warn_internal_delegate_pkce_if_applicable(new_server, source="database")
|
||||
await self._persist_discovered_obo_token_url(
|
||||
server_id=mcp_server.server_id,
|
||||
auth_type=auth_type,
|
||||
existing_token_url=mcp_server.token_url,
|
||||
discovered_token_url=new_server.token_url,
|
||||
)
|
||||
return new_server
|
||||
|
||||
async def _persist_discovered_obo_token_url(
|
||||
self,
|
||||
*,
|
||||
server_id: str,
|
||||
auth_type: Optional[MCPAuthType],
|
||||
existing_token_url: Optional[str],
|
||||
discovered_token_url: Optional[str],
|
||||
) -> None:
|
||||
"""Write a freshly discovered OBO token endpoint back onto the DB row.
|
||||
|
||||
``build_mcp_server_from_table`` resolves ``token_url`` via RFC 9728 -> RFC 8414 for an
|
||||
``oauth2_token_exchange`` server that has none configured, but that resolved value otherwise
|
||||
lives only on the returned in-memory object; the row keeps ``token_url=None`` so every rebuild
|
||||
re-runs discovery, and a transient upstream outage during a rebuild leaves the server with no
|
||||
endpoint until discovery next succeeds. Persisting it makes ``_obo_needs_endpoint_discovery``
|
||||
return False on the next build. Fires at most once per server (skipped once the row has a
|
||||
value), and is best-effort: a write failure just means discovery runs again next time.
|
||||
"""
|
||||
if auth_type != MCPAuth.oauth2_token_exchange:
|
||||
return
|
||||
if existing_token_url or not discovered_token_url:
|
||||
return
|
||||
from litellm.proxy.proxy_server import prisma_client # noqa: PLC0415
|
||||
|
||||
if prisma_client is None:
|
||||
return
|
||||
try:
|
||||
await MCPServerRepository(prisma_client).table.update(
|
||||
where={"server_id": server_id},
|
||||
data={"token_url": discovered_token_url},
|
||||
)
|
||||
verbose_logger.debug("Persisted discovered OBO token_url for MCP server %s", server_id)
|
||||
except Exception as exc: # noqa: BLE001 - best-effort; a failed write re-discovers next build
|
||||
verbose_logger.warning("Failed to persist discovered OBO token_url for MCP server %s: %s", server_id, exc)
|
||||
|
||||
async def _maybe_register_openapi_tools(self, server: MCPServer, *, initialize_mapping: bool = True):
|
||||
"""Register OpenAPI tools if the server has a spec_path configured."""
|
||||
if server.spec_path:
|
||||
|
|
@ -1668,6 +1751,21 @@ class MCPServerManager:
|
|||
return auth_value
|
||||
return None
|
||||
|
||||
def _obo_subject_token(
|
||||
self,
|
||||
server: MCPServer,
|
||||
raw_headers: Optional[dict[str, str]],
|
||||
) -> Optional[str]:
|
||||
"""The caller's bearer as the token_exchange (OBO) subject token, for that mode only.
|
||||
|
||||
Prompts/resources discovery and reads on a token_exchange server must exchange the caller's
|
||||
token like the tools paths do, not connect with no credential. Other modes never read the
|
||||
inbound bearer, so return None to avoid forwarding it.
|
||||
"""
|
||||
if server.auth_type != MCPAuth.oauth2_token_exchange:
|
||||
return None
|
||||
return self._extract_bearer_token(None, raw_headers)
|
||||
|
||||
def _build_stdio_env(
|
||||
self,
|
||||
server: MCPServer,
|
||||
|
|
@ -1858,6 +1956,84 @@ class MCPServerManager:
|
|||
_write_user_env_vars_cache(user_id, server.server_id, values)
|
||||
return values
|
||||
|
||||
async def _resolve_v2_auth(
|
||||
self,
|
||||
*,
|
||||
server: MCPServer,
|
||||
spec: ServerSpec,
|
||||
provider: UpstreamCredentialProvider,
|
||||
subject_token: Optional[str],
|
||||
user_api_key_auth: Optional[UserAPIKeyAuth],
|
||||
extra_headers: Optional[dict[str, str]],
|
||||
) -> tuple[Optional[httpx.Auth], Optional[dict[str, str]]]:
|
||||
"""Resolve a v2-owned server's upstream credential into ``(resolved_auth, extra_headers)``.
|
||||
|
||||
On a missing/rejected per-user credential this raises the mode's discovery challenge
|
||||
(authorization_code's browser-OAuth 401, token_exchange's RFC 9728 challenge) or maps any
|
||||
other ``CredError`` onto its public HTTP status; it never returns an error as a value.
|
||||
"""
|
||||
match await provider.resolve_credentials(to_subject(user_api_key_auth, subject_token), spec):
|
||||
case Ok(auth):
|
||||
# NoOpAuth has no header_name and so never conflicts.
|
||||
header_name = getattr(auth, "header_name", None)
|
||||
conflicts = bool(
|
||||
header_name and extra_headers and any(key.lower() == header_name.lower() for key in extra_headers)
|
||||
)
|
||||
if not conflicts:
|
||||
return auth, extra_headers
|
||||
if isinstance(spec.config, (TokenExchangeConfig, AuthorizationCodeConfig)):
|
||||
# The resolver owns the per-user credential here (token_exchange's exchanged
|
||||
# token, authorization_code's stored token). It is authoritative: a guardrail such
|
||||
# as MCPJWTSigner, static_headers, or any other injected Authorization must NOT
|
||||
# shadow it (otherwise the upstream gets e.g. the signer's JWT instead of the
|
||||
# exchanged token and rejects it). Drop the conflicting header so the resolved
|
||||
# token reaches upstream.
|
||||
return auth, _without_authorization(extra_headers)
|
||||
# Other modes: an Authorization already supplied via extra_headers (a forwarded caller
|
||||
# header or static_headers) is intentional and wins; v1 applies those last.
|
||||
return None, extra_headers
|
||||
case Error(err):
|
||||
if err.tag == "unauthorized" and isinstance(spec.config, AuthorizationCodeConfig):
|
||||
# authorization_code's missing per-user token -> the per-server browser-OAuth
|
||||
# challenge, built here where the full MCPServer is in hand.
|
||||
raise_user_oauth_challenge(server, root_path=get_server_root_path())
|
||||
if err.tag == "unauthorized" and isinstance(spec.config, TokenExchangeConfig):
|
||||
# token_exchange (OBO): a missing/rejected subject token -> the RFC 9728 challenge
|
||||
# pointing at the IdP the client must SSO with to obtain one, rather than an opaque
|
||||
# 401. No gateway-side browser flow.
|
||||
raise_token_exchange_challenge(server, root_path=get_server_root_path())
|
||||
raise_public(err)
|
||||
|
||||
async def preflight_token_exchange(
|
||||
self,
|
||||
server: MCPServer,
|
||||
oauth2_headers: Optional[dict[str, str]],
|
||||
user_api_key_auth: Optional[UserAPIKeyAuth],
|
||||
) -> None:
|
||||
"""Run the OBO exchange for a caller-supplied subject at the transport edge.
|
||||
|
||||
Single-server routes call this before the MCP session opens, where an HTTP status and
|
||||
``WWW-Authenticate`` still reach the client. A rejected subject raises the RFC 9728
|
||||
challenge and any other ``CredError`` maps onto its public HTTP status, so an exchange
|
||||
failure surfaces as a failure instead of the session continuing into an empty tool list.
|
||||
A successful exchange is cached by the exchanger, so the session's list/call reuses it.
|
||||
"""
|
||||
if server.auth_type != MCPAuth.oauth2_token_exchange:
|
||||
return
|
||||
subject_token = self._extract_bearer_token(oauth2_headers, None)
|
||||
if not subject_token:
|
||||
return
|
||||
spec = to_server_spec(server)
|
||||
if spec is None or not isinstance(spec.config, TokenExchangeConfig):
|
||||
return
|
||||
match await self._cred_provider.resolve_credentials(to_subject(user_api_key_auth, subject_token), spec):
|
||||
case Ok(_):
|
||||
return
|
||||
case Error(err):
|
||||
if err.tag == "unauthorized":
|
||||
raise_token_exchange_challenge(server, root_path=get_server_root_path())
|
||||
raise_public(err)
|
||||
|
||||
async def _create_mcp_client(
|
||||
self,
|
||||
server: MCPServer,
|
||||
|
|
@ -1892,11 +2068,17 @@ class MCPServerManager:
|
|||
spec = None if transport == MCPTransport.stdio else to_server_spec(server)
|
||||
provider = cred_provider or self._cred_provider
|
||||
# A caller-supplied per-request override (mcp_auth_header / x-mcp-*) defers to the v1 path
|
||||
# so it wins - except for authorization_code, whose per-user token the v2 resolver owns. A
|
||||
# caller must not be able to substitute another user's stored credential, so we keep the v2
|
||||
# spec and ignore the override there; the REST tools preview supplies its not-yet-persisted
|
||||
# token through the resolver (cred_provider), never this path.
|
||||
if spec is not None and mcp_auth_header and not isinstance(spec.config, AuthorizationCodeConfig):
|
||||
# so it wins - except for the per-user modes the v2 resolver owns (authorization_code's
|
||||
# stored token and token_exchange's RFC 8693 minted token). A caller must not be able to
|
||||
# substitute another user's stored credential, nor silently disable the OBO exchange and
|
||||
# forward an arbitrary bearer upstream, so we keep the v2 spec and ignore the override for
|
||||
# both; the REST tools preview supplies its not-yet-persisted token through the resolver
|
||||
# (cred_provider), never this path.
|
||||
if (
|
||||
spec is not None
|
||||
and mcp_auth_header
|
||||
and not isinstance(spec.config, (AuthorizationCodeConfig, TokenExchangeConfig))
|
||||
):
|
||||
spec = None
|
||||
auth_value = (
|
||||
await resolve_mcp_auth(server, mcp_auth_header, subject_token=subject_token) if spec is None else None
|
||||
|
|
@ -1962,26 +2144,14 @@ class MCPServerManager:
|
|||
server_url = server.url or ""
|
||||
|
||||
if spec is not None:
|
||||
match await provider.resolve_credentials(to_subject(user_api_key_auth, subject_token), spec):
|
||||
case Ok(auth):
|
||||
resolved_auth = auth
|
||||
# Do not override an Authorization already supplied via extra_headers
|
||||
# (a guardrail hook such as the JWT signer, static_headers, or a
|
||||
# forwarded caller header): v1 applies those last, so they win. NoOpAuth
|
||||
# has no header_name and so never skips.
|
||||
header_name = getattr(resolved_auth, "header_name", None)
|
||||
if (
|
||||
header_name
|
||||
and extra_headers
|
||||
and any(key.lower() == header_name.lower() for key in extra_headers)
|
||||
):
|
||||
resolved_auth = None
|
||||
case Error(err):
|
||||
if err.tag == "unauthorized":
|
||||
# The arm signals a missing per-user token semantically; raise the
|
||||
# per-server OAuth challenge here, where the full MCPServer is in hand.
|
||||
raise_user_oauth_challenge(server)
|
||||
raise_public(err)
|
||||
resolved_auth, extra_headers = await self._resolve_v2_auth(
|
||||
server=server,
|
||||
spec=spec,
|
||||
provider=provider,
|
||||
subject_token=subject_token,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
extra_headers=extra_headers,
|
||||
)
|
||||
return MCPClient(
|
||||
server_url=server_url,
|
||||
transport_type=transport,
|
||||
|
|
@ -2026,6 +2196,7 @@ class MCPServerManager:
|
|||
add_prefix: bool = True,
|
||||
raw_headers: Optional[dict[str, str]] = None,
|
||||
user_api_key_auth: Optional[UserAPIKeyAuth] = None,
|
||||
oauth2_headers: Optional[dict[str, str]] = None,
|
||||
) -> list[MCPTool]:
|
||||
"""
|
||||
Helper method to get tools from a single MCP server with prefixed names.
|
||||
|
|
@ -2099,11 +2270,21 @@ class MCPServerManager:
|
|||
|
||||
stdio_env = self._build_stdio_env(server, raw_headers)
|
||||
|
||||
# token_exchange (OBO) discovery needs the caller's token too: list it with the user's own
|
||||
# token (mirrors the call path), not v1's deleted client_credentials fallback. Other modes
|
||||
# never read the inbound bearer, so leave subject_token None to avoid forwarding it.
|
||||
subject_token = (
|
||||
self._extract_bearer_token(oauth2_headers, raw_headers)
|
||||
if server.auth_type == MCPAuth.oauth2_token_exchange
|
||||
else None
|
||||
)
|
||||
|
||||
client = await self._create_mcp_client(
|
||||
server=server,
|
||||
mcp_auth_header=mcp_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
|
||||
|
|
@ -2143,12 +2324,16 @@ class MCPServerManager:
|
|||
# aggregator catches this explicitly to keep absorbing.
|
||||
raise
|
||||
except HTTPException as e:
|
||||
headers = e.headers or {}
|
||||
www_authenticate = headers.get("WWW-Authenticate") or headers.get("www-authenticate")
|
||||
if e.status_code == 401 and www_authenticate is not None:
|
||||
# A v2 resolver auth challenge (token_exchange's RFC 9728 401, authorization_code's
|
||||
# browser-OAuth 401, or a 403) is raised at client-build time, inside this try. Route it
|
||||
# through the same MCPUpstreamAuthError channel as pass-through so single-server routes
|
||||
# surface the challenge (the client re-authenticates) while the aggregator keeps absorbing.
|
||||
# Non-auth HTTP errors stay absorbed so one misconfigured server can't blank the listing.
|
||||
if e.status_code in (401, 403):
|
||||
headers = e.headers or {}
|
||||
raise MCPUpstreamAuthError(
|
||||
status_code=401,
|
||||
www_authenticate=www_authenticate,
|
||||
status_code=e.status_code,
|
||||
www_authenticate=headers.get("WWW-Authenticate") or headers.get("www-authenticate"),
|
||||
server_name=server.name,
|
||||
) from e
|
||||
verbose_logger.warning(f"Failed to get tools from server {server.name}: {str(e)}")
|
||||
|
|
@ -2188,12 +2373,14 @@ class MCPServerManager:
|
|||
extra_headers.update(server.static_headers)
|
||||
|
||||
stdio_env = self._build_stdio_env(server, raw_headers)
|
||||
subject_token = self._obo_subject_token(server, raw_headers)
|
||||
|
||||
client = await self._create_mcp_client(
|
||||
server=server,
|
||||
mcp_auth_header=mcp_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
)
|
||||
|
||||
prompts = await client.list_prompts()
|
||||
|
|
@ -2228,12 +2415,14 @@ class MCPServerManager:
|
|||
extra_headers.update(server.static_headers)
|
||||
|
||||
stdio_env = self._build_stdio_env(server, raw_headers)
|
||||
subject_token = self._obo_subject_token(server, raw_headers)
|
||||
|
||||
client = await self._create_mcp_client(
|
||||
server=server,
|
||||
mcp_auth_header=mcp_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
)
|
||||
|
||||
resources = await client.list_resources()
|
||||
|
|
@ -2268,12 +2457,14 @@ class MCPServerManager:
|
|||
extra_headers.update(server.static_headers)
|
||||
|
||||
stdio_env = self._build_stdio_env(server, raw_headers)
|
||||
subject_token = self._obo_subject_token(server, raw_headers)
|
||||
|
||||
client = await self._create_mcp_client(
|
||||
server=server,
|
||||
mcp_auth_header=mcp_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
)
|
||||
|
||||
resource_templates = await client.list_resource_templates()
|
||||
|
|
@ -2307,12 +2498,14 @@ class MCPServerManager:
|
|||
extra_headers.update(server.static_headers)
|
||||
|
||||
stdio_env = self._build_stdio_env(server, raw_headers)
|
||||
subject_token = self._obo_subject_token(server, raw_headers)
|
||||
|
||||
client = await self._create_mcp_client(
|
||||
server=server,
|
||||
mcp_auth_header=mcp_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
)
|
||||
|
||||
return await client.read_resource(url)
|
||||
|
|
@ -2337,12 +2530,14 @@ class MCPServerManager:
|
|||
extra_headers.update(server.static_headers)
|
||||
|
||||
stdio_env = self._build_stdio_env(server, raw_headers)
|
||||
subject_token = self._obo_subject_token(server, raw_headers)
|
||||
|
||||
client = await self._create_mcp_client(
|
||||
server=server,
|
||||
mcp_auth_header=mcp_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
)
|
||||
|
||||
get_prompt_request_params = GetPromptRequestParams(
|
||||
|
|
@ -2395,8 +2590,17 @@ class MCPServerManager:
|
|||
async def _descovery_metadata(
|
||||
self,
|
||||
server_url: str,
|
||||
*,
|
||||
allow_origin_fallback: bool = True,
|
||||
) -> Optional[MCPOAuthMetadata]:
|
||||
"""Discover OAuth metadata by following RFC 9728 (protected resource metadata discovery)."""
|
||||
"""Discover OAuth metadata by following RFC 9728 (protected resource metadata discovery).
|
||||
|
||||
``allow_origin_fallback`` controls the last-resort guess that treats the resource server's own
|
||||
origin as its authorization server when nothing is advertised. The browser ``oauth2`` flow keeps
|
||||
it (a human sees the redirect), but token_exchange (OBO) sets it False so the gateway never
|
||||
exchanges a subject token against an endpoint it inferred rather than one explicitly configured
|
||||
or authoritatively advertised via RFC 9728 / RFC 8414.
|
||||
"""
|
||||
|
||||
try:
|
||||
client = get_async_httpx_client(llm_provider=httpxSpecialProvider.MCP)
|
||||
|
|
@ -2446,7 +2650,7 @@ class MCPServerManager:
|
|||
) = await self._attempt_well_known_discovery(server_url)
|
||||
|
||||
metadata = None
|
||||
if not authorization_servers:
|
||||
if allow_origin_fallback and not authorization_servers:
|
||||
try:
|
||||
parsed_url = urlparse(server_url)
|
||||
if parsed_url.scheme and parsed_url.netloc:
|
||||
|
|
@ -2608,6 +2812,14 @@ class MCPServerManager:
|
|||
continue
|
||||
|
||||
scopes = self._extract_scopes(data.get("scopes_supported"))
|
||||
verbose_logger.debug(
|
||||
"Authorization server metadata from %s: issuer=%s grant_types_supported=%s "
|
||||
"token_endpoint_auth_methods_supported=%s",
|
||||
url,
|
||||
data.get("issuer"),
|
||||
data.get("grant_types_supported"),
|
||||
data.get("token_endpoint_auth_methods_supported"),
|
||||
)
|
||||
metadata = MCPOAuthMetadata(
|
||||
scopes=scopes,
|
||||
authorization_url=data.get("authorization_endpoint"),
|
||||
|
|
@ -3236,6 +3448,46 @@ class MCPServerManager:
|
|||
async with semaphore:
|
||||
yield
|
||||
|
||||
async def _obo_call_tool_with_retry(
|
||||
self,
|
||||
*,
|
||||
client: MCPClient,
|
||||
call_tool_params: MCPCallToolRequestParams,
|
||||
host_progress_callback: Optional[Callable],
|
||||
mcp_server: MCPServer,
|
||||
server_auth_header: str | dict[str, str] | None,
|
||||
extra_headers: Optional[dict[str, str]],
|
||||
stdio_env: Optional[dict[str, str]],
|
||||
subject_token: Optional[str],
|
||||
user_api_key_auth: Optional[UserAPIKeyAuth],
|
||||
) -> CallToolResult:
|
||||
"""Call a token_exchange (OBO) tool; on an upstream 401/403 re-mint the token once and retry.
|
||||
|
||||
The exchanged token is baked into the client at build time, so the retry invalidates the
|
||||
cached exchange and rebuilds the client (which re-exchanges). One retry only: a non-auth
|
||||
failure or a second auth failure degrades to the normal ``isError`` result, and a re-exchange
|
||||
that now fails surfaces its own 401 challenge from ``_create_mcp_client``.
|
||||
"""
|
||||
try:
|
||||
return await client.call_tool(
|
||||
call_tool_params, host_progress_callback=host_progress_callback, raise_on_error=True
|
||||
)
|
||||
except Exception as exc:
|
||||
if _extract_upstream_auth_failure(exc) is None:
|
||||
return MCPClient.error_tool_result(exc)
|
||||
spec = to_server_spec(mcp_server)
|
||||
if spec is not None:
|
||||
await self._cred_provider.invalidate_credentials(to_subject(user_api_key_auth, subject_token), spec)
|
||||
retry_client = await self._create_mcp_client(
|
||||
server=mcp_server,
|
||||
mcp_auth_header=server_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
return await retry_client.call_tool(call_tool_params, host_progress_callback=host_progress_callback)
|
||||
|
||||
async def _call_regular_mcp_tool(
|
||||
self,
|
||||
mcp_server: MCPServer,
|
||||
|
|
@ -3394,11 +3646,30 @@ class MCPServerManager:
|
|||
arguments=arguments,
|
||||
)
|
||||
|
||||
async def _call_tool_via_client(client, params):
|
||||
async with self._limit_outbound_concurrency(mcp_server):
|
||||
return await client.call_tool(params, host_progress_callback=host_progress_callback)
|
||||
if mcp_server.auth_type == MCPAuth.oauth2_token_exchange and subject_token:
|
||||
# OBO: the exchanged token may have been revoked/rotated upstream since it was cached, so
|
||||
# an upstream 401 gets one re-mint + retry. Gated to this mode; all others keep the plain
|
||||
# single call below.
|
||||
tool_call_coro = self._obo_call_tool_with_retry(
|
||||
client=client,
|
||||
call_tool_params=call_tool_params,
|
||||
host_progress_callback=host_progress_callback,
|
||||
mcp_server=mcp_server,
|
||||
server_auth_header=server_auth_header,
|
||||
extra_headers=extra_headers,
|
||||
stdio_env=stdio_env,
|
||||
subject_token=subject_token,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
else:
|
||||
|
||||
tasks.append(asyncio.create_task(_call_tool_via_client(client, call_tool_params)))
|
||||
async def _call_tool_via_client(client, params):
|
||||
async with self._limit_outbound_concurrency(mcp_server):
|
||||
return await client.call_tool(params, host_progress_callback=host_progress_callback)
|
||||
|
||||
tool_call_coro = _call_tool_via_client(client, call_tool_params)
|
||||
|
||||
tasks.append(asyncio.create_task(tool_call_coro))
|
||||
|
||||
_timeout = mcp_server.timeout if mcp_server.timeout is not None else MCP_CLIENT_TIMEOUT
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ from litellm.proxy._experimental.mcp_server.outbound_credentials.types import (
|
|||
ServerSpec,
|
||||
SharedKey,
|
||||
Subject,
|
||||
TokenExchangeConfig,
|
||||
)
|
||||
from litellm.types.mcp import MCPAuth
|
||||
|
||||
|
|
@ -61,8 +62,9 @@ def to_server_spec(server: MCPServer) -> Optional[ServerSpec]:
|
|||
an ``assert_never`` tail, so a newly added auth mode fails the type gate here until it is
|
||||
explicitly mapped or explicitly deferred, rather than silently falling through to v1. Live
|
||||
modes: ``none``, the static-header family (``api_key`` plus the Authorization schemes,
|
||||
all shared-key), and ``oauth2`` per-user tokens (``authorization_code``); client_credentials
|
||||
(M2M), delegated/passthrough oauth2, token exchange, and SigV4 return None and stay on v1.
|
||||
all shared-key), ``oauth2`` per-user tokens (``authorization_code``), and
|
||||
``oauth2_token_exchange`` (RFC 8693 OBO); client_credentials (M2M), delegated/passthrough
|
||||
oauth2, and SigV4 return None and stay on v1.
|
||||
"""
|
||||
if server.is_byok:
|
||||
return None # per-user BYOK source not migrated yet -> defer to v1 (any auth_type)
|
||||
|
|
@ -92,11 +94,41 @@ def to_server_spec(server: MCPServer) -> Optional[ServerSpec]:
|
|||
)
|
||||
# client_credentials (M2M) and delegate/passthrough oauth2 stay on v1
|
||||
return None
|
||||
case MCPAuth.oauth2_token_exchange | MCPAuth.aws_sigv4:
|
||||
return None # token exchange and SigV4 are not migrated yet -> defer to v1
|
||||
case MCPAuth.oauth2_token_exchange:
|
||||
return _token_exchange_spec(server, resource)
|
||||
case MCPAuth.aws_sigv4:
|
||||
return None # SigV4 is not migrated yet -> defer to v1
|
||||
assert_never(auth_type)
|
||||
|
||||
|
||||
def _token_exchange_spec(server: MCPServer, resource: str) -> Optional[ServerSpec]:
|
||||
"""Build a token_exchange (RFC 8693 OBO) spec, or defer (None) when it is not OBO-configured.
|
||||
|
||||
An OBO server with ``client_id``/``client_secret`` is owned by the v2 arm even if the
|
||||
``token_exchange_endpoint``/``token_url`` is absent: a missing endpoint then fails closed (412) at
|
||||
the exchanger rather than silently deferring to v1 and connecting unauthenticated, since the
|
||||
gateway must not guess the IdP or fall back to a weaker source. Without client credentials there is
|
||||
nothing to own, so the server stays on v1 (parity-safe). ``audience`` is forwarded only when the
|
||||
operator set it; a missing one is omitted, not derived.
|
||||
"""
|
||||
endpoint = server.token_exchange_endpoint or server.token_url
|
||||
if not server.client_id or not server.client_secret:
|
||||
return None
|
||||
return ServerSpec(
|
||||
server_id=server.server_id,
|
||||
resource=resource,
|
||||
config=TokenExchangeConfig(
|
||||
subject_token_type=server.subject_token_type or "urn:ietf:params:oauth:token-type:access_token",
|
||||
token_exchange_endpoint=endpoint,
|
||||
audience=server.audience,
|
||||
client_id=server.client_id,
|
||||
client_secret=SecretStr(server.client_secret),
|
||||
token_endpoint_auth_method=server.token_endpoint_auth_method,
|
||||
scopes=tuple(server.scopes or ()),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _shared_key_spec(
|
||||
server: MCPServer,
|
||||
resource: str,
|
||||
|
|
@ -148,23 +180,52 @@ def raise_public(error: CredError) -> NoReturn:
|
|||
assert_never(error.tag)
|
||||
|
||||
|
||||
def raise_user_oauth_challenge(server: MCPServer) -> NoReturn:
|
||||
def oauth_protected_resource_path(root_path: str, server: MCPServer) -> str:
|
||||
"""The server's RFC 9728 Protected Resource Metadata path, the shared anchor of both challenges.
|
||||
|
||||
``root_path`` is the proxy's ``SERVER_ROOT_PATH``, resolved by the caller (the imperative shell)
|
||||
so this stays a pure function of its inputs; ``"/"`` and ``""`` both mean no prefix. The path is
|
||||
relative, so it resolves against the caller's own host (correct even behind a reverse proxy).
|
||||
"""
|
||||
prefix = "" if root_path == "/" else root_path
|
||||
name = server.alias or server.server_name or server.name or server.server_id
|
||||
return f"/.well-known/oauth-protected-resource{prefix}/mcp/{name}"
|
||||
|
||||
|
||||
def raise_user_oauth_challenge(server: MCPServer, *, root_path: str) -> NoReturn:
|
||||
"""Raise the 401 an ``authorization_code`` server returns at egress when the user has no token.
|
||||
|
||||
Points at the server's RFC 9728 Protected Resource Metadata (``resource_metadata``), which names
|
||||
the upstream authorization server the client must complete OAuth with. The URL is per-server and
|
||||
relative, so it resolves against the caller's own host (correct even behind a reverse proxy)
|
||||
without needing request context. The listing-phase 401 still emits the RFC 8414 ``authorization_uri``
|
||||
form pending the format unification; both target the same server, so the difference is cosmetic.
|
||||
Points at the server's RFC 9728 Protected Resource Metadata, which names the upstream
|
||||
authorization server the client must complete OAuth with. The listing-phase 401 still emits the
|
||||
RFC 8414 ``authorization_uri`` form pending the format unification; both target the same server,
|
||||
so the difference is cosmetic.
|
||||
"""
|
||||
from litellm.proxy.utils import get_server_root_path # noqa: PLC0415
|
||||
|
||||
root = get_server_root_path()
|
||||
prefix = "" if root == "/" else root
|
||||
name = server.alias or server.server_name or server.name or server.server_id
|
||||
resource_metadata = f"/.well-known/oauth-protected-resource{prefix}/mcp/{name}"
|
||||
resource_metadata = oauth_protected_resource_path(root_path, server)
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail="Unauthorized",
|
||||
headers={"WWW-Authenticate": f'Bearer resource_metadata="{resource_metadata}"'},
|
||||
)
|
||||
|
||||
|
||||
def raise_token_exchange_challenge(server: MCPServer, *, root_path: str) -> NoReturn:
|
||||
"""Raise the RFC 9728 / RFC 6750 challenge an OBO (``token_exchange``) server returns when the
|
||||
caller's subject token is missing or the IdP rejected it.
|
||||
|
||||
Points at the server's Protected Resource Metadata, whose ``authorization_servers`` names the IdP
|
||||
the client must SSO with to obtain a subject token; ``error="invalid_token"`` tells a
|
||||
spec-compliant MCP client to discover that AS and retry with a fresh bearer. Mirrors
|
||||
``raise_user_oauth_challenge`` but for the exchange flow: there is no gateway-side browser OAuth —
|
||||
the client re-authenticates directly with the IdP, and LiteLLM then exchanges the resulting token.
|
||||
"""
|
||||
resource_metadata = oauth_protected_resource_path(root_path, server)
|
||||
www_authenticate = (
|
||||
f'Bearer resource_metadata="{resource_metadata}", '
|
||||
'error="invalid_token", '
|
||||
'error_description="Missing or invalid subject token; authenticate with the IdP and retry"'
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail="Unauthorized",
|
||||
headers={"WWW-Authenticate": www_authenticate},
|
||||
)
|
||||
|
|
|
|||
|
|
@ -8,7 +8,8 @@ an arm fails the type gate (basedpyright `reportMatchNotExhaustive`); a bypassed
|
|||
at runtime instead of returning `None`.
|
||||
|
||||
`none` and `api_key` (shared-key source) are live, as is `authorization_code`, which reads the
|
||||
user's token from the injected `OAuthTokenStore`. The remaining arms are `not_implemented` stubs
|
||||
user's token from the injected `OAuthTokenStore`, and `token_exchange`, which swaps the caller's
|
||||
inbound token through the injected `TokenExchanger`. The remaining arms are `not_implemented` stubs
|
||||
that each land in a follow-up PR with their seam. Pure v2: no imports from v1.
|
||||
"""
|
||||
|
||||
|
|
@ -31,6 +32,9 @@ from litellm.proxy._experimental.mcp_server.outbound_credentials.result import (
|
|||
Ok,
|
||||
Result,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.token_exchanger import (
|
||||
TokenExchanger,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.types import (
|
||||
ApiKeyConfig,
|
||||
AuthorizationCodeConfig,
|
||||
|
|
@ -55,16 +59,36 @@ class _NullOAuthTokenStore:
|
|||
return None
|
||||
|
||||
|
||||
class _NullTokenExchanger:
|
||||
"""Fail-closed default: with no exchanger wired, token_exchange cannot produce a credential."""
|
||||
|
||||
async def exchange(
|
||||
self, subject_token: str, server: ServerSpec, config: TokenExchangeConfig, *, tenant_id: str = ""
|
||||
) -> Result[OAuthToken, CredError]:
|
||||
return Error(CredError.of_misconfigured("token exchange collaborator not wired"))
|
||||
|
||||
async def invalidate(
|
||||
self, subject_token: str, server: ServerSpec, config: TokenExchangeConfig, *, tenant_id: str = ""
|
||||
) -> None:
|
||||
return None
|
||||
|
||||
|
||||
class UpstreamCredentialProvider:
|
||||
"""Produces the one `httpx.Auth` for a `(subject, upstream)` pair, per declared mode.
|
||||
|
||||
Collaborators (the per-mode credential stores and token fetchers) are injected as each arm is
|
||||
built; the live `none` and `api_key`-shared arms read from the config and need none, while
|
||||
`authorization_code` reads the user's token from the injected `OAuthTokenStore`.
|
||||
`authorization_code` reads the user's token from the injected `OAuthTokenStore` and
|
||||
`token_exchange` swaps the caller's token through the injected `TokenExchanger`.
|
||||
"""
|
||||
|
||||
def __init__(self, oauth_token_store: OAuthTokenStore | None = None) -> None:
|
||||
def __init__(
|
||||
self,
|
||||
oauth_token_store: OAuthTokenStore | None = None,
|
||||
token_exchanger: TokenExchanger | None = None,
|
||||
) -> None:
|
||||
self._oauth_token_store: OAuthTokenStore = oauth_token_store or _NullOAuthTokenStore()
|
||||
self._token_exchanger: TokenExchanger = token_exchanger or _NullTokenExchanger()
|
||||
|
||||
async def resolve_credentials(self, subject: Subject, server: ServerSpec) -> Result[httpx.Auth, CredError]:
|
||||
match server.config:
|
||||
|
|
@ -76,8 +100,8 @@ class UpstreamCredentialProvider:
|
|||
return _not_implemented(AuthSpecKind.passthrough)
|
||||
case ClientCredentialsConfig():
|
||||
return _not_implemented(AuthSpecKind.client_credentials)
|
||||
case TokenExchangeConfig():
|
||||
return _not_implemented(AuthSpecKind.token_exchange)
|
||||
case TokenExchangeConfig() as config:
|
||||
return await self._token_exchange(subject, server, config)
|
||||
case AuthorizationCodeConfig():
|
||||
return await self._authorization_code(subject, server)
|
||||
case AwsSigV4Config():
|
||||
|
|
@ -110,6 +134,43 @@ class UpstreamCredentialProvider:
|
|||
return Error(CredError.of_unauthorized("Authorization required: complete the OAuth flow for this server."))
|
||||
return Ok(StaticHeaderAuth(f"Bearer {token.access_token}", header_name="Authorization"))
|
||||
|
||||
async def _token_exchange(
|
||||
self, subject: Subject, server: ServerSpec, config: TokenExchangeConfig
|
||||
) -> Result[StaticHeaderAuth, CredError]:
|
||||
"""RFC 8693 OBO: exchange the caller's inbound token for an upstream-bound bearer.
|
||||
|
||||
No inbound token means there is nothing to exchange, so the arm fails closed with a 401 rather
|
||||
than falling through to a weaker source (§1.5); the exchanger handles the IdP round-trip and
|
||||
caching and returns the upstream token or a typed error.
|
||||
"""
|
||||
inbound = subject.inbound_token
|
||||
if inbound is None:
|
||||
return Error(
|
||||
CredError.of_unauthorized(
|
||||
"Token exchange requires a caller token to exchange (OBO).",
|
||||
www_authenticate='Bearer error="invalid_request"',
|
||||
)
|
||||
)
|
||||
match await self._token_exchanger.exchange(
|
||||
inbound.get_secret_value(), server, config, tenant_id=subject.tenant_id
|
||||
):
|
||||
case Ok(token):
|
||||
return Ok(StaticHeaderAuth(f"Bearer {token.access_token}", header_name="Authorization"))
|
||||
case Error(err):
|
||||
return Error(err)
|
||||
|
||||
async def invalidate_credentials(self, subject: Subject, server: ServerSpec) -> None:
|
||||
"""Drop any cached credential the resolver owns for this `(subject, server)`.
|
||||
|
||||
Used after an upstream rejects the injected credential, so the next resolve re-mints rather
|
||||
than serving the same rejected token until TTL. Only `token_exchange` holds a re-mintable
|
||||
cached credential here; other modes are a no-op.
|
||||
"""
|
||||
if isinstance(server.config, TokenExchangeConfig) and subject.inbound_token is not None:
|
||||
await self._token_exchanger.invalidate(
|
||||
subject.inbound_token.get_secret_value(), server, server.config, tenant_id=subject.tenant_id
|
||||
)
|
||||
|
||||
async def _authz_token(self, subject: Subject, server: ServerSpec) -> OAuthToken | None:
|
||||
"""The user's authorization_code token, or None when absent or the store is unreachable.
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,105 @@
|
|||
"""Composition root for the v2-native token_exchange (OBO) exchanger.
|
||||
|
||||
Wires the pure ``Rfc8693TokenExchanger`` to its runtime edges: the real httpx POST against the IdP and
|
||||
the configured cache sizing/TTL constants. ``build_token_exchanger`` is built once at egress
|
||||
construction and reused, so the in-process exchanged-token cache survives across requests. Unlike the
|
||||
per-user store, nothing here reads a runtime global at build time (the httpx client is acquired per
|
||||
call), so it needs no lazy wrapper.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import (
|
||||
MCP_OAUTH2_TOKEN_CACHE_DEFAULT_TTL,
|
||||
MCP_OAUTH2_TOKEN_CACHE_MIN_TTL,
|
||||
MCP_OAUTH2_TOKEN_EXPIRY_BUFFER_SECONDS,
|
||||
MCP_TOKEN_EXCHANGE_CACHE_MAX_SIZE,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.oauth_token_store import (
|
||||
InMemoryTokenCacheBackend,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.token_exchanger import (
|
||||
Rfc8693TokenExchanger,
|
||||
SubjectTokenRejected,
|
||||
TokenExchangeClientError,
|
||||
)
|
||||
|
||||
# RFC 6749 5.2 error codes that mean the gateway's own request/credentials are wrong (not the
|
||||
# caller's subject token), so they surface as a 500 the caller can't fix by re-authenticating.
|
||||
_GATEWAY_FAULT_OAUTH_ERRORS = frozenset(
|
||||
{"invalid_client", "unauthorized_client", "unsupported_grant_type", "invalid_target", "invalid_scope"}
|
||||
)
|
||||
|
||||
|
||||
def _oauth_error_code(response: httpx.Response) -> str | None:
|
||||
"""Read the RFC 6749 5.2 ``error`` code from a token-endpoint error body, or None if absent.
|
||||
|
||||
The ``error_description`` is deliberately not read: it can carry IdP internals and must never
|
||||
reach the caller. Only the standard machine code drives classification.
|
||||
"""
|
||||
try:
|
||||
body: object = response.json()
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
if isinstance(body, dict):
|
||||
code = body.get("error")
|
||||
if isinstance(code, str):
|
||||
return code
|
||||
return None
|
||||
|
||||
|
||||
async def _post_exchange_endpoint(
|
||||
url: str, form: dict[str, str], client_auth_headers: dict[str, str]
|
||||
) -> dict[str, object] | None:
|
||||
from litellm.llms.custom_httpx.http_handler import ( # noqa: PLC0415
|
||||
get_async_httpx_client, # pyright: ignore
|
||||
)
|
||||
from litellm.types.llms.custom_http import httpxSpecialProvider # noqa: PLC0415
|
||||
|
||||
# litellm's httpx handler and httpx.Response are only partially typed; the IdP returns a JSON
|
||||
# object and the exchanger validates each field, so the untyped boundary is contained here.
|
||||
# A 4xx is the IdP rejecting the subject (non-retryable -> 401 via SubjectTokenRejected); any
|
||||
# other failure is a miss (-> None -> upstream_unavailable -> 503), matching v1's fail-closed.
|
||||
headers = {"Accept": "application/json", **client_auth_headers}
|
||||
try:
|
||||
client = get_async_httpx_client(llm_provider=httpxSpecialProvider.MCP) # pyright: ignore
|
||||
response = await client.post(url, headers=headers, data=form) # pyright: ignore
|
||||
response.raise_for_status() # pyright: ignore
|
||||
parsed: object = response.json() # pyright: ignore
|
||||
except httpx.HTTPStatusError as status_err:
|
||||
status_code = status_err.response.status_code
|
||||
if 400 <= status_code < 500:
|
||||
oauth_error = _oauth_error_code(status_err.response)
|
||||
if oauth_error in _GATEWAY_FAULT_OAUTH_ERRORS:
|
||||
verbose_logger.warning(
|
||||
"MCP token exchange rejected as %s (HTTP %d); check the gateway client credentials, "
|
||||
"audience, and scope for this server",
|
||||
oauth_error,
|
||||
status_code,
|
||||
)
|
||||
raise TokenExchangeClientError(oauth_error) from status_err
|
||||
raise SubjectTokenRejected(f"IdP rejected the subject token (HTTP {status_code})") from status_err
|
||||
verbose_logger.warning("MCP token exchange request failed: %s", status_err)
|
||||
return None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
verbose_logger.warning("MCP token exchange request failed: %s", exc)
|
||||
return None
|
||||
if not isinstance(parsed, dict):
|
||||
# A valid-but-non-object JSON body (list/string/number) would crash the field parsing; map it
|
||||
# to a miss so it surfaces as a typed upstream_unavailable, not a 500.
|
||||
verbose_logger.warning("MCP token exchange returned non-object JSON (%s)", type(parsed).__name__)
|
||||
return None
|
||||
return parsed # pyright: ignore
|
||||
|
||||
|
||||
def build_token_exchanger() -> Rfc8693TokenExchanger:
|
||||
return Rfc8693TokenExchanger(
|
||||
_post_exchange_endpoint,
|
||||
cache=InMemoryTokenCacheBackend(max_size=MCP_TOKEN_EXCHANGE_CACHE_MAX_SIZE),
|
||||
default_ttl_seconds=MCP_OAUTH2_TOKEN_CACHE_DEFAULT_TTL,
|
||||
min_ttl_seconds=MCP_OAUTH2_TOKEN_CACHE_MIN_TTL,
|
||||
expiry_buffer_seconds=MCP_OAUTH2_TOKEN_EXPIRY_BUFFER_SECONDS,
|
||||
)
|
||||
|
|
@ -0,0 +1,298 @@
|
|||
"""v2-native RFC 8693 token exchange (OBO): swap the caller's token for an upstream one.
|
||||
|
||||
The pure core of the ``token_exchange`` mode. Given the caller's ``subject_token`` and the server's
|
||||
``TokenExchangeConfig``, ``Rfc8693TokenExchanger.exchange`` POSTs the RFC 8693 token-exchange grant to
|
||||
the configured endpoint and returns the upstream-bound ``access_token`` as a typed ``OAuthToken``, or a
|
||||
typed ``CredError`` - never a raise (the HTTP edge is the injected ``ExchangeHttpPost``, whose adapter
|
||||
contains the I/O). The exchanged token is cached and single-flighted per ``(subject_token, server)`` so
|
||||
a repeated caller token skips the IdP round-trip and concurrent calls collapse to one exchange, reusing
|
||||
the shared in-process cache + coordinator foundation. A rotated caller token hashes to a new key and
|
||||
re-exchanges. Pure v2 apart from the shared RFC 6749 client-auth helper, which carries no v1 state.
|
||||
|
||||
A missing/expired exchange is an error, never a fall-through to a weaker source (§1.5): the caller
|
||||
presenting no token is the resolver arm's 401, and an IdP that does not return a usable token is an
|
||||
``upstream_unavailable`` here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import time
|
||||
from collections.abc import Awaitable, Callable
|
||||
from typing import Protocol
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.oauth_token_store import (
|
||||
InMemoryTokenCacheBackend,
|
||||
InProcessRefreshCoordinator,
|
||||
OAuthToken,
|
||||
RefreshCoordinator,
|
||||
TokenCacheBackend,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.result import (
|
||||
Error,
|
||||
Ok,
|
||||
Result,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.auth.token_endpoint_auth import (
|
||||
build_token_endpoint_client_auth,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.types import (
|
||||
CredError,
|
||||
ServerSpec,
|
||||
TokenExchangeConfig,
|
||||
)
|
||||
|
||||
# A token with no declared expiry is cached for this long; one with an expiry is cached until then
|
||||
# minus the skew buffer, floored at the minimum. Values mirror v1's MCP_OAUTH2_* constants; the
|
||||
# composition root injects the configured ones.
|
||||
_DEFAULT_TTL_SECONDS = 3600.0
|
||||
_MIN_TTL_SECONDS = 10.0
|
||||
_EXPIRY_BUFFER_SECONDS = 60.0
|
||||
|
||||
_GRANT_TYPE = "urn:ietf:params:oauth:grant-type:token-exchange"
|
||||
|
||||
# RFC 8693 3 token-type URNs that are not usable as an upstream Bearer access token. token_type
|
||||
# already rejects the common non-access case (N_A); this catches a malformed STS that mints one of
|
||||
# these but still labels it Bearer. An access_token / jwt / absent / unknown type is accepted (lenient).
|
||||
_NON_ACCESS_ISSUED_TOKEN_TYPES = frozenset(
|
||||
{
|
||||
"urn:ietf:params:oauth:token-type:refresh_token",
|
||||
"urn:ietf:params:oauth:token-type:id_token",
|
||||
"urn:ietf:params:oauth:token-type:saml1",
|
||||
"urn:ietf:params:oauth:token-type:saml2",
|
||||
}
|
||||
)
|
||||
|
||||
# The IdP returns an opaque JSON object; the post adapter hands it over untyped and the exchanger
|
||||
# validates each field, so no Any leaks past this seam (None == any transport/HTTP failure). The
|
||||
# second dict is the form body; the third is the client-auth headers (HTTP Basic for
|
||||
# client_secret_basic, empty for client_secret_post).
|
||||
ExchangeHttpPost = Callable[[str, "dict[str, str]", "dict[str, str]"], Awaitable["dict[str, object] | None"]]
|
||||
|
||||
|
||||
class SubjectTokenRejected(Exception):
|
||||
"""The IdP refused to exchange the subject token (an RFC 8693 4xx, e.g. ``invalid_grant``).
|
||||
|
||||
Distinct from a transport / IdP-availability failure, which the post adapter maps to ``None`` ->
|
||||
``upstream_unavailable`` -> 503 (retryable). A rejected subject is the caller's problem, not the
|
||||
gateway's, so the arm surfaces it as a non-retryable 401 (the OBO challenge) instead.
|
||||
"""
|
||||
|
||||
|
||||
class TokenExchangeClientError(Exception):
|
||||
"""The IdP rejected the exchange for a reason that is the gateway's fault, not the caller's.
|
||||
|
||||
RFC 6749 5.2 codes such as ``invalid_client`` (the gateway's own STS credentials are wrong),
|
||||
``unauthorized_client`` / ``unsupported_grant_type`` (the gateway is not permitted to exchange),
|
||||
``invalid_target`` / ``invalid_scope`` (the gateway's audience/scope config for this server is
|
||||
wrong). The caller cannot fix these by re-authenticating, so the arm surfaces them as a 500
|
||||
(``misconfigured``), not the 401 OBO challenge. The IdP ``error_description`` is never carried.
|
||||
"""
|
||||
|
||||
|
||||
class TokenExchanger(Protocol):
|
||||
"""Exchanges a caller token for an upstream-bound one, per the server's token_exchange config."""
|
||||
|
||||
async def exchange(
|
||||
self, subject_token: str, server: ServerSpec, config: TokenExchangeConfig, *, tenant_id: str = ""
|
||||
) -> Result[OAuthToken, CredError]: ...
|
||||
|
||||
async def invalidate(
|
||||
self, subject_token: str, server: ServerSpec, config: TokenExchangeConfig, *, tenant_id: str = ""
|
||||
) -> None: ...
|
||||
|
||||
|
||||
def _cache_key(subject_token: str, tenant_id: str, config: TokenExchangeConfig) -> str:
|
||||
"""Bind the cache entry to the caller token, the tenant, AND the exchange config that minted it.
|
||||
|
||||
A rotated caller token, a different tenant, endpoint, audience, scope, client_id, secret, auth
|
||||
method, or subject_token_type all change the key, so two tenants behind the same opaque token
|
||||
never share an entry and a config change forces a fresh exchange instead of serving a token
|
||||
minted for the old config until TTL. Everything is hashed, so no secret is held in the key.
|
||||
"""
|
||||
secret = config.client_secret.get_secret_value() if config.client_secret else ""
|
||||
material = "\x00".join(
|
||||
(
|
||||
subject_token,
|
||||
tenant_id,
|
||||
config.token_exchange_endpoint or "",
|
||||
config.audience or "",
|
||||
config.subject_token_type,
|
||||
config.client_id or "",
|
||||
secret,
|
||||
config.token_endpoint_auth_method or "",
|
||||
" ".join(config.scopes),
|
||||
)
|
||||
)
|
||||
return hashlib.sha256(material.encode()).hexdigest()
|
||||
|
||||
|
||||
def _parse_expires_in(raw: object) -> int | None:
|
||||
if isinstance(raw, bool):
|
||||
return None
|
||||
if isinstance(raw, (int, float)):
|
||||
return int(raw)
|
||||
if isinstance(raw, str):
|
||||
try:
|
||||
return int(float(raw))
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _build_exchange_form(
|
||||
*,
|
||||
subject_token: str,
|
||||
subject_token_type: str,
|
||||
audience: str | None,
|
||||
scopes: tuple[str, ...],
|
||||
) -> dict[str, str]:
|
||||
return {
|
||||
"grant_type": _GRANT_TYPE,
|
||||
"subject_token": subject_token,
|
||||
"subject_token_type": subject_token_type,
|
||||
**({"audience": audience} if audience else {}),
|
||||
**({"scope": " ".join(scopes)} if scopes else {}),
|
||||
}
|
||||
|
||||
|
||||
class Rfc8693TokenExchanger:
|
||||
"""``TokenExchanger`` that runs the RFC 8693 grant once per caller token, then caches the result.
|
||||
|
||||
The HTTP post is injected (``None`` on any IdP failure, mirroring v1: a failed exchange is a miss,
|
||||
not a 500). The cache and single-flight coordinator default to the in-process foundation; a
|
||||
deployment with no shared state needs nothing more (v1's exchanged-token cache is per-process too).
|
||||
The clock is injected so TTL/expiry is deterministic in tests.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
http_post: ExchangeHttpPost,
|
||||
*,
|
||||
cache: TokenCacheBackend | None = None,
|
||||
coordinator: RefreshCoordinator | None = None,
|
||||
clock: Callable[[], float] = time.time,
|
||||
default_ttl_seconds: float = _DEFAULT_TTL_SECONDS,
|
||||
min_ttl_seconds: float = _MIN_TTL_SECONDS,
|
||||
expiry_buffer_seconds: float = _EXPIRY_BUFFER_SECONDS,
|
||||
) -> None:
|
||||
self._http_post = http_post
|
||||
self._cache: TokenCacheBackend = cache or InMemoryTokenCacheBackend(clock=clock)
|
||||
self._coordinator: RefreshCoordinator = coordinator or InProcessRefreshCoordinator()
|
||||
self._clock = clock
|
||||
self._default_ttl_seconds = default_ttl_seconds
|
||||
self._min_ttl_seconds = min_ttl_seconds
|
||||
self._expiry_buffer_seconds = expiry_buffer_seconds
|
||||
|
||||
async def exchange(
|
||||
self, subject_token: str, server: ServerSpec, config: TokenExchangeConfig, *, tenant_id: str = ""
|
||||
) -> Result[OAuthToken, CredError]:
|
||||
endpoint = config.token_exchange_endpoint
|
||||
client_id = config.client_id
|
||||
client_secret = config.client_secret
|
||||
if not endpoint:
|
||||
# No endpoint configured and none discoverable: fail closed (412) rather than guess an IdP
|
||||
# or fall back to a weaker source. The caller's token is never sent anywhere.
|
||||
return Error(
|
||||
CredError.of_precondition_required("token exchange endpoint is not configured for this server")
|
||||
)
|
||||
if not client_id or client_secret is None:
|
||||
return Error(CredError.of_misconfigured("token_exchange requires client_id and client_secret"))
|
||||
|
||||
cache_key = _cache_key(subject_token, tenant_id, config)
|
||||
server_id = server.server_id
|
||||
cached = await self._cache.get(cache_key, server_id)
|
||||
if cached is not None:
|
||||
verbose_logger.debug("MCP token exchange cache hit for server %s", server_id)
|
||||
return Ok(cached)
|
||||
|
||||
client_auth = build_token_endpoint_client_auth(
|
||||
auth_method=config.token_endpoint_auth_method,
|
||||
client_id=client_id,
|
||||
client_secret=client_secret.get_secret_value(),
|
||||
)
|
||||
form = {
|
||||
**_build_exchange_form(
|
||||
subject_token=subject_token,
|
||||
subject_token_type=config.subject_token_type,
|
||||
audience=config.audience,
|
||||
scopes=config.scopes,
|
||||
),
|
||||
**client_auth.body,
|
||||
}
|
||||
|
||||
async def run_exchange() -> OAuthToken | None:
|
||||
fresh = await self._cache.get(cache_key, server_id)
|
||||
if fresh is not None:
|
||||
return fresh
|
||||
verbose_logger.debug(
|
||||
"Exchanging token for MCP server %s at %s (audience=%s)", server_id, endpoint, config.audience
|
||||
)
|
||||
body = await self._http_post(endpoint, form, client_auth.headers)
|
||||
if body is None:
|
||||
return None
|
||||
token = self._token_from_body(body)
|
||||
if token is None:
|
||||
return None
|
||||
await self._cache.set(cache_key, server_id, token, self._ttl_seconds(token))
|
||||
verbose_logger.info("Token exchange succeeded for MCP server %s", server_id)
|
||||
return token
|
||||
|
||||
async def reread() -> OAuthToken | None:
|
||||
return await self._cache.get(cache_key, server_id)
|
||||
|
||||
try:
|
||||
token = await self._coordinator.run(cache_key, server_id, refresh=run_exchange, reread=reread)
|
||||
except SubjectTokenRejected as rejected:
|
||||
# The IdP rejected the subject token (4xx). This is non-retryable: the caller must
|
||||
# re-authenticate with the IdP, so it surfaces as a 401 (the OBO challenge), not a 503.
|
||||
return Error(CredError.of_unauthorized(str(rejected) or "subject token rejected by the IdP"))
|
||||
except TokenExchangeClientError:
|
||||
# RFC 6749 5.2 gateway-fault code (invalid_client / invalid_target / ...): the caller can't
|
||||
# fix it by re-authenticating, so surface a 500 rather than the OBO 401 challenge. The
|
||||
# specific code is logged at the edge; the user-facing summary stays generic.
|
||||
return Error(
|
||||
CredError.of_misconfigured(
|
||||
"token exchange configuration error: the gateway's credentials, audience, or scope "
|
||||
"for this server were not accepted by the IdP"
|
||||
)
|
||||
)
|
||||
if token is None:
|
||||
return Error(CredError.of_upstream_unavailable("token exchange did not return a usable access token"))
|
||||
return Ok(token)
|
||||
|
||||
async def invalidate(
|
||||
self, subject_token: str, server: ServerSpec, config: TokenExchangeConfig, *, tenant_id: str = ""
|
||||
) -> None:
|
||||
"""Drop the cached exchanged token so the next call re-exchanges (e.g. after an upstream 401)."""
|
||||
await self._cache.delete(_cache_key(subject_token, tenant_id, config), server.server_id)
|
||||
|
||||
def _token_from_body(self, body: dict[str, object]) -> OAuthToken | None:
|
||||
access_token = body.get("access_token")
|
||||
if not isinstance(access_token, str) or not access_token:
|
||||
return None
|
||||
# token_type is forwarded downstream as Bearer, so a present-but-non-Bearer type (e.g. N_A)
|
||||
# must fail closed rather than be minted as a bogus Bearer; an absent type defaults to Bearer.
|
||||
token_type = body.get("token_type")
|
||||
if isinstance(token_type, str) and token_type.strip().lower() != "bearer":
|
||||
verbose_logger.warning(
|
||||
"MCP token exchange returned unusable token_type %r; refusing to forward it as Bearer", token_type
|
||||
)
|
||||
return None
|
||||
# issued_token_type says what representation was minted; reject a clearly-non-access type
|
||||
# (refresh/id/saml) even if token_type claimed Bearer. access_token / jwt / absent / unknown pass.
|
||||
issued_token_type = body.get("issued_token_type")
|
||||
if isinstance(issued_token_type, str) and issued_token_type in _NON_ACCESS_ISSUED_TOKEN_TYPES:
|
||||
return None
|
||||
expires_in = _parse_expires_in(body.get("expires_in"))
|
||||
expires_at = self._clock() + expires_in if expires_in is not None else None
|
||||
return OAuthToken(access_token=access_token, expires_at=expires_at)
|
||||
|
||||
def _ttl_seconds(self, token: OAuthToken) -> float:
|
||||
if token.expires_at is None:
|
||||
return self._default_ttl_seconds
|
||||
lifetime = max(0.0, token.expires_at - self._clock())
|
||||
# Floor at min_ttl, but never cache past the token's own expiry: a token whose remaining
|
||||
# lifetime is below the buffer (or even below min_ttl) must not be served stale upstream.
|
||||
return min(max(lifetime - self._expiry_buffer_seconds, self._min_ttl_seconds), lifetime)
|
||||
|
|
@ -183,17 +183,23 @@ class ClientCredentialsConfig(BaseModel):
|
|||
|
||||
class TokenExchangeConfig(BaseModel):
|
||||
"""RFC 8693 OBO; swap the caller's live subject_token for a token bound to the upstream's
|
||||
audience (`server.resource`, RFC 8707). The gateway authenticates to the exchange endpoint
|
||||
as an OAuth client (`client_id`/`client_secret`); the inbound token is sent only to that
|
||||
endpoint, never to the upstream.
|
||||
audience. The gateway authenticates to the exchange endpoint as an OAuth client
|
||||
(`client_id`/`client_secret`); the inbound token is sent only to that endpoint, never to the
|
||||
upstream.
|
||||
|
||||
`audience` is the RFC 8693 target; it is optional and sent only when the operator configured
|
||||
one, since both `audience` and `resource` are optional in the spec and the authorization server
|
||||
applies its own default when neither is sent (fabricating one risks `invalid_target`).
|
||||
"""
|
||||
|
||||
model_config = ConfigDict(frozen=True)
|
||||
kind: Literal[AuthSpecKind.token_exchange] = AuthSpecKind.token_exchange
|
||||
subject_token_type: str = "urn:ietf:params:oauth:token-type:access_token"
|
||||
token_exchange_endpoint: str | None = None
|
||||
audience: str | None = None
|
||||
client_id: str | None = None
|
||||
client_secret: SecretStr | None = None
|
||||
token_endpoint_auth_method: Literal["client_secret_basic", "client_secret_post"] | None = None
|
||||
scopes: tuple[str, ...] = ()
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1838,6 +1838,7 @@ if MCP_AVAILABLE:
|
|||
add_prefix=True, # Always add server prefix
|
||||
raw_headers=raw_headers,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
oauth2_headers=oauth2_headers,
|
||||
)
|
||||
filtered_tools = filter_tools_by_allowed_tools(tools, server)
|
||||
|
||||
|
|
@ -1860,7 +1861,8 @@ if MCP_AVAILABLE:
|
|||
# tools. Surfacing the upstream 401 to the client as a re-auth challenge is
|
||||
# intentionally not done here: raising from this list handler cannot produce a
|
||||
# 401 + WWW-Authenticate (the MCP session manager serializes it as a JSON-RPC
|
||||
# error), so that belongs in a request-scope preemptive check, tracked separately.
|
||||
# error). Single-server routes surface it via the request-scope preemptive
|
||||
# check in _raise_preemptive_401_for_unauthenticated_servers instead.
|
||||
verbose_logger.debug(f"MCP list_tools: omitting {server.name}; it needs upstream auth")
|
||||
return []
|
||||
except Exception as e:
|
||||
|
|
@ -2692,12 +2694,18 @@ if MCP_AVAILABLE:
|
|||
|
||||
# Forward named client headers to OpenAPI tool upstream requests.
|
||||
# MCPServer.extra_headers lists header names to copy from raw_headers.
|
||||
# OAuth2 M2M: never take Authorization from the caller (matches
|
||||
# _prepare_mcp_server_headers for managed MCP).
|
||||
# The strip decision is centralized in _should_strip_caller_authorization so this
|
||||
# OpenAPI/local path agrees with the managed paths: M2M and the resolver-owned modes
|
||||
# (token_exchange's raw subject token, authorization_code's stored token) must never
|
||||
# have the caller's Authorization forwarded verbatim upstream.
|
||||
forwarded_headers: Optional[Dict[str, str]] = None
|
||||
if mcp_server and mcp_server.extra_headers and raw_headers:
|
||||
normalized_raw = {str(k).lower(): v for k, v in raw_headers.items() if isinstance(k, str)}
|
||||
skip_caller_authorization = bool(mcp_server.has_client_credentials)
|
||||
skip_caller_authorization = _should_strip_caller_authorization(
|
||||
mcp_server=mcp_server,
|
||||
raw_headers=raw_headers,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
for header_name in mcp_server.extra_headers:
|
||||
if not isinstance(header_name, str):
|
||||
continue
|
||||
|
|
@ -3466,6 +3474,36 @@ if MCP_AVAILABLE:
|
|||
headers={"www-authenticate": authorization_uri},
|
||||
)
|
||||
|
||||
# token_exchange (OBO): the caller supplied no subject token. Challenge at connect
|
||||
# (transport level, where WWW-Authenticate survives) with the RFC 9728 resource_metadata
|
||||
# so the client discovers the IdP, SSOs, and retries with a subject token, which LiteLLM
|
||||
# then exchanges. A tool-call-time 401 would be wrapped into a JSON-RPC error and the
|
||||
# header lost, so the discovery flow needs this pre-emptive challenge.
|
||||
if server and server.auth_type == MCPAuth.oauth2_token_exchange and not oauth2_headers:
|
||||
from litellm.proxy._experimental.mcp_server.outbound_credentials.adapter import ( # noqa: PLC0415
|
||||
raise_token_exchange_challenge,
|
||||
)
|
||||
from litellm.proxy.utils import get_server_root_path # noqa: PLC0415
|
||||
|
||||
raise_token_exchange_challenge(server, root_path=get_server_root_path())
|
||||
|
||||
# token_exchange (OBO) with a subject present: run the exchange here at the transport
|
||||
# edge, so a rejected subject raises the RFC 9728 challenge (and a gateway fault its
|
||||
# public status) instead of the session opening and list_tools masking the failure as
|
||||
# an empty tool list. Gated to single-server routes; the multi-server aggregate keeps
|
||||
# absorbing per-server auth failures so one bad server cannot 401 the whole connect.
|
||||
if (
|
||||
server
|
||||
and server.auth_type == MCPAuth.oauth2_token_exchange
|
||||
and oauth2_headers
|
||||
and len(mcp_servers or []) == 1
|
||||
):
|
||||
await global_mcp_server_manager.preflight_token_exchange(
|
||||
server=server,
|
||||
oauth2_headers=oauth2_headers,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
|
||||
# Pass-through OAuth: when the admin has opted a server into
|
||||
# forwarding the client's bearer token (is_oauth_passthrough) and
|
||||
# the client hasn't supplied one, fail fast with 401 and point
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -1,9 +1,9 @@
|
|||
1:"$Sreact.fragment"
|
||||
2:I[347257,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"ClientPageRoot"]
|
||||
3:I[871135,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js","/litellm-asset-prefix/_next/static/chunks/0whkizop7gd0~.js","/litellm-asset-prefix/_next/static/chunks/0-ih8xcz_89nt.js","/litellm-asset-prefix/_next/static/chunks/0pd5zl~lciww9.js","/litellm-asset-prefix/_next/static/chunks/02ihc5xweq16v.js","/litellm-asset-prefix/_next/static/chunks/0lg.6rbfsd-l9.js","/litellm-asset-prefix/_next/static/chunks/0mzw3maijoev6.js","/litellm-asset-prefix/_next/static/chunks/043q3g5-5-aju.js","/litellm-asset-prefix/_next/static/chunks/04amwk-x_vjxu.js","/litellm-asset-prefix/_next/static/chunks/0-dhh1_d1.b1u.js","/litellm-asset-prefix/_next/static/chunks/0pwkd9r.mc_ee.js","/litellm-asset-prefix/_next/static/chunks/011mgw.-67gs_.js","/litellm-asset-prefix/_next/static/chunks/0~-ovi6c4wjt1.js","/litellm-asset-prefix/_next/static/chunks/0_y-b9_d9dsuv.js","/litellm-asset-prefix/_next/static/chunks/0c2apcdkbqq0o.js","/litellm-asset-prefix/_next/static/chunks/0zrbitbm~0koh.js","/litellm-asset-prefix/_next/static/chunks/0sx3mu2_l9g_y.js","/litellm-asset-prefix/_next/static/chunks/0ngre0.s4-ej6.js","/litellm-asset-prefix/_next/static/chunks/0l7em-5kjv49e.js","/litellm-asset-prefix/_next/static/chunks/05t1k89l9tc3s.js","/litellm-asset-prefix/_next/static/chunks/17n.qg70cy9.9.js","/litellm-asset-prefix/_next/static/chunks/00q4mtjboprhm.js","/litellm-asset-prefix/_next/static/chunks/0el08tticy_20.js","/litellm-asset-prefix/_next/static/chunks/0-3i_.uof35pm.js","/litellm-asset-prefix/_next/static/chunks/14566-_ogh-19.js","/litellm-asset-prefix/_next/static/chunks/0w39dn9x3dp9g.js","/litellm-asset-prefix/_next/static/chunks/0q6~n4y84cejn.js","/litellm-asset-prefix/_next/static/chunks/0v1rxqc1hqmrl.js","/litellm-asset-prefix/_next/static/chunks/0c4pfjjue0uc-.js"],"default"]
|
||||
6:I[897367,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"OutletBoundary"]
|
||||
2:I[347257,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"ClientPageRoot"]
|
||||
3:I[871135,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js","/litellm-asset-prefix/_next/static/chunks/0uuigwiz-in3~.js","/litellm-asset-prefix/_next/static/chunks/0-0c4mv4-mc9n.js","/litellm-asset-prefix/_next/static/chunks/0lg.6rbfsd-l9.js","/litellm-asset-prefix/_next/static/chunks/0p3v32gvsxp6h.js","/litellm-asset-prefix/_next/static/chunks/0x09ws363q4_0.js","/litellm-asset-prefix/_next/static/chunks/08o64zaid_juv.js","/litellm-asset-prefix/_next/static/chunks/043q3g5-5-aju.js","/litellm-asset-prefix/_next/static/chunks/0g9k1~ppf2hw3.js","/litellm-asset-prefix/_next/static/chunks/151r-5htw45m~.js","/litellm-asset-prefix/_next/static/chunks/0v85l0arelm41.js","/litellm-asset-prefix/_next/static/chunks/0zbgu4ogb6mba.js","/litellm-asset-prefix/_next/static/chunks/0ae3np_qb52e-.js","/litellm-asset-prefix/_next/static/chunks/0zam.8alu6_vj.js","/litellm-asset-prefix/_next/static/chunks/0uu6lckpr0s15.js","/litellm-asset-prefix/_next/static/chunks/0.bx44y-6~tug.js","/litellm-asset-prefix/_next/static/chunks/0l7em-5kjv49e.js","/litellm-asset-prefix/_next/static/chunks/0el08tticy_20.js","/litellm-asset-prefix/_next/static/chunks/0-s2am3eulbyd.js","/litellm-asset-prefix/_next/static/chunks/0sx3mu2_l9g_y.js","/litellm-asset-prefix/_next/static/chunks/0.w8~sa9q0n_s.js","/litellm-asset-prefix/_next/static/chunks/00q4mtjboprhm.js","/litellm-asset-prefix/_next/static/chunks/0zrbitbm~0koh.js","/litellm-asset-prefix/_next/static/chunks/0c4pfjjue0uc-.js","/litellm-asset-prefix/_next/static/chunks/0efmbzvj03niy.js","/litellm-asset-prefix/_next/static/chunks/055egae-ggkjh.js","/litellm-asset-prefix/_next/static/chunks/0hwip5a7qsmis.js","/litellm-asset-prefix/_next/static/chunks/0q6~n4y84cejn.js","/litellm-asset-prefix/_next/static/chunks/0mh1wnrvmv_y7.js"],"default"]
|
||||
6:I[897367,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"OutletBoundary"]
|
||||
7:"$Sreact.suspense"
|
||||
0:{"rsc":["$","$1","c",{"children":[["$","$L2",null,{"Component":"$3","serverProvidedParams":{"searchParams":{},"params":{},"promises":["$@4","$@5"]}}],[["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/011mgw.-67gs_.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/0~-ovi6c4wjt1.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/0_y-b9_d9dsuv.js","async":true}],["$","script","script-3",{"src":"/litellm-asset-prefix/_next/static/chunks/0c2apcdkbqq0o.js","async":true}],["$","script","script-4",{"src":"/litellm-asset-prefix/_next/static/chunks/0zrbitbm~0koh.js","async":true}],["$","script","script-5",{"src":"/litellm-asset-prefix/_next/static/chunks/0sx3mu2_l9g_y.js","async":true}],["$","script","script-6",{"src":"/litellm-asset-prefix/_next/static/chunks/0ngre0.s4-ej6.js","async":true}],["$","script","script-7",{"src":"/litellm-asset-prefix/_next/static/chunks/0l7em-5kjv49e.js","async":true}],["$","script","script-8",{"src":"/litellm-asset-prefix/_next/static/chunks/05t1k89l9tc3s.js","async":true}],["$","script","script-9",{"src":"/litellm-asset-prefix/_next/static/chunks/17n.qg70cy9.9.js","async":true}],["$","script","script-10",{"src":"/litellm-asset-prefix/_next/static/chunks/00q4mtjboprhm.js","async":true}],["$","script","script-11",{"src":"/litellm-asset-prefix/_next/static/chunks/0el08tticy_20.js","async":true}],["$","script","script-12",{"src":"/litellm-asset-prefix/_next/static/chunks/0-3i_.uof35pm.js","async":true}],["$","script","script-13",{"src":"/litellm-asset-prefix/_next/static/chunks/14566-_ogh-19.js","async":true}],["$","script","script-14",{"src":"/litellm-asset-prefix/_next/static/chunks/0w39dn9x3dp9g.js","async":true}],["$","script","script-15",{"src":"/litellm-asset-prefix/_next/static/chunks/0q6~n4y84cejn.js","async":true}],["$","script","script-16",{"src":"/litellm-asset-prefix/_next/static/chunks/0v1rxqc1hqmrl.js","async":true}],["$","script","script-17",{"src":"/litellm-asset-prefix/_next/static/chunks/0c4pfjjue0uc-.js","async":true}]],["$","$L6",null,{"children":["$","$7",null,{"name":"Next.MetadataOutlet","children":"$@8"}]}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"5rDiFx0t_mOGYmV_8kSkw"}
|
||||
0:{"rsc":["$","$1","c",{"children":[["$","$L2",null,{"Component":"$3","serverProvidedParams":{"searchParams":{},"params":{},"promises":["$@4","$@5"]}}],[["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/0ae3np_qb52e-.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/0zam.8alu6_vj.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/0uu6lckpr0s15.js","async":true}],["$","script","script-3",{"src":"/litellm-asset-prefix/_next/static/chunks/0.bx44y-6~tug.js","async":true}],["$","script","script-4",{"src":"/litellm-asset-prefix/_next/static/chunks/0l7em-5kjv49e.js","async":true}],["$","script","script-5",{"src":"/litellm-asset-prefix/_next/static/chunks/0el08tticy_20.js","async":true}],["$","script","script-6",{"src":"/litellm-asset-prefix/_next/static/chunks/0-s2am3eulbyd.js","async":true}],["$","script","script-7",{"src":"/litellm-asset-prefix/_next/static/chunks/0sx3mu2_l9g_y.js","async":true}],["$","script","script-8",{"src":"/litellm-asset-prefix/_next/static/chunks/0.w8~sa9q0n_s.js","async":true}],["$","script","script-9",{"src":"/litellm-asset-prefix/_next/static/chunks/00q4mtjboprhm.js","async":true}],["$","script","script-10",{"src":"/litellm-asset-prefix/_next/static/chunks/0zrbitbm~0koh.js","async":true}],["$","script","script-11",{"src":"/litellm-asset-prefix/_next/static/chunks/0c4pfjjue0uc-.js","async":true}],["$","script","script-12",{"src":"/litellm-asset-prefix/_next/static/chunks/0efmbzvj03niy.js","async":true}],["$","script","script-13",{"src":"/litellm-asset-prefix/_next/static/chunks/055egae-ggkjh.js","async":true}],["$","script","script-14",{"src":"/litellm-asset-prefix/_next/static/chunks/0hwip5a7qsmis.js","async":true}],["$","script","script-15",{"src":"/litellm-asset-prefix/_next/static/chunks/0q6~n4y84cejn.js","async":true}],["$","script","script-16",{"src":"/litellm-asset-prefix/_next/static/chunks/0mh1wnrvmv_y7.js","async":true}]],["$","$L6",null,{"children":["$","$7",null,{"name":"Next.MetadataOutlet","children":"$@8"}]}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"KYqiq5stbD-H4YcZ-6OuP"}
|
||||
4:{}
|
||||
5:"$0:rsc:props:children:0:props:serverProvidedParams:params"
|
||||
8:null
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
1:"$Sreact.fragment"
|
||||
2:I[92825,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"ClientSegmentRoot"]
|
||||
3:I[216370,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js","/litellm-asset-prefix/_next/static/chunks/0whkizop7gd0~.js","/litellm-asset-prefix/_next/static/chunks/0-ih8xcz_89nt.js","/litellm-asset-prefix/_next/static/chunks/0pd5zl~lciww9.js","/litellm-asset-prefix/_next/static/chunks/02ihc5xweq16v.js","/litellm-asset-prefix/_next/static/chunks/0lg.6rbfsd-l9.js","/litellm-asset-prefix/_next/static/chunks/0mzw3maijoev6.js","/litellm-asset-prefix/_next/static/chunks/043q3g5-5-aju.js","/litellm-asset-prefix/_next/static/chunks/04amwk-x_vjxu.js","/litellm-asset-prefix/_next/static/chunks/0-dhh1_d1.b1u.js","/litellm-asset-prefix/_next/static/chunks/0pwkd9r.mc_ee.js"],"default"]
|
||||
4:I[339756,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
5:I[837457,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
0:{"rsc":["$","$1","c",{"children":[[["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/0whkizop7gd0~.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/0-ih8xcz_89nt.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/0pd5zl~lciww9.js","async":true}],["$","script","script-3",{"src":"/litellm-asset-prefix/_next/static/chunks/02ihc5xweq16v.js","async":true}],["$","script","script-4",{"src":"/litellm-asset-prefix/_next/static/chunks/0lg.6rbfsd-l9.js","async":true}],["$","script","script-5",{"src":"/litellm-asset-prefix/_next/static/chunks/0mzw3maijoev6.js","async":true}],["$","script","script-6",{"src":"/litellm-asset-prefix/_next/static/chunks/043q3g5-5-aju.js","async":true}],["$","script","script-7",{"src":"/litellm-asset-prefix/_next/static/chunks/04amwk-x_vjxu.js","async":true}],["$","script","script-8",{"src":"/litellm-asset-prefix/_next/static/chunks/0-dhh1_d1.b1u.js","async":true}],["$","script","script-9",{"src":"/litellm-asset-prefix/_next/static/chunks/0pwkd9r.mc_ee.js","async":true}]],["$","$L2",null,{"Component":"$3","slots":{"children":["$","$L4",null,{"parallelRouterKey":"children","template":["$","$L5",null,{}],"notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]]}]},"serverProvidedParams":{"params":{},"promises":["$@6"]}}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"5rDiFx0t_mOGYmV_8kSkw"}
|
||||
2:I[92825,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"ClientSegmentRoot"]
|
||||
3:I[216370,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js","/litellm-asset-prefix/_next/static/chunks/0uuigwiz-in3~.js","/litellm-asset-prefix/_next/static/chunks/0-0c4mv4-mc9n.js","/litellm-asset-prefix/_next/static/chunks/0lg.6rbfsd-l9.js","/litellm-asset-prefix/_next/static/chunks/0p3v32gvsxp6h.js","/litellm-asset-prefix/_next/static/chunks/0x09ws363q4_0.js","/litellm-asset-prefix/_next/static/chunks/08o64zaid_juv.js","/litellm-asset-prefix/_next/static/chunks/043q3g5-5-aju.js","/litellm-asset-prefix/_next/static/chunks/0g9k1~ppf2hw3.js","/litellm-asset-prefix/_next/static/chunks/151r-5htw45m~.js","/litellm-asset-prefix/_next/static/chunks/0v85l0arelm41.js","/litellm-asset-prefix/_next/static/chunks/0zbgu4ogb6mba.js"],"default"]
|
||||
4:I[339756,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
5:I[837457,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
0:{"rsc":["$","$1","c",{"children":[[["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/0uuigwiz-in3~.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/0-0c4mv4-mc9n.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/0lg.6rbfsd-l9.js","async":true}],["$","script","script-3",{"src":"/litellm-asset-prefix/_next/static/chunks/0p3v32gvsxp6h.js","async":true}],["$","script","script-4",{"src":"/litellm-asset-prefix/_next/static/chunks/0x09ws363q4_0.js","async":true}],["$","script","script-5",{"src":"/litellm-asset-prefix/_next/static/chunks/08o64zaid_juv.js","async":true}],["$","script","script-6",{"src":"/litellm-asset-prefix/_next/static/chunks/043q3g5-5-aju.js","async":true}],["$","script","script-7",{"src":"/litellm-asset-prefix/_next/static/chunks/0g9k1~ppf2hw3.js","async":true}],["$","script","script-8",{"src":"/litellm-asset-prefix/_next/static/chunks/151r-5htw45m~.js","async":true}],["$","script","script-9",{"src":"/litellm-asset-prefix/_next/static/chunks/0v85l0arelm41.js","async":true}],["$","script","script-10",{"src":"/litellm-asset-prefix/_next/static/chunks/0zbgu4ogb6mba.js","async":true}]],["$","$L2",null,{"Component":"$3","slots":{"children":["$","$L4",null,{"parallelRouterKey":"children","template":["$","$L5",null,{}],"notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]]}]},"serverProvidedParams":{"params":{},"promises":["$@6"]}}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"KYqiq5stbD-H4YcZ-6OuP"}
|
||||
6:"$0:rsc:props:children:1:props:serverProvidedParams:params"
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,6 +1,6 @@
|
|||
1:"$Sreact.fragment"
|
||||
2:I[897367,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"ViewportBoundary"]
|
||||
3:I[897367,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"MetadataBoundary"]
|
||||
2:I[897367,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"ViewportBoundary"]
|
||||
3:I[897367,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"MetadataBoundary"]
|
||||
4:"$Sreact.suspense"
|
||||
5:I[27201,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"IconMark"]
|
||||
0:{"rsc":["$","$1","h",{"children":[null,["$","$L2",null,{"children":[["$","meta","0",{"charSet":"utf-8"}],["$","meta","1",{"name":"viewport","content":"width=device-width, initial-scale=1"}]]}],["$","div",null,{"hidden":true,"children":["$","$L3",null,{"children":["$","$4",null,{"name":"Next.Metadata","children":[["$","title","0",{"children":"LiteLLM Dashboard"}],["$","meta","1",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","2",{"rel":"icon","href":"/favicon.ico?favicon.0~dgapwhi~75y.ico","sizes":"48x48","type":"image/x-icon"}],["$","link","3",{"rel":"icon","href":"/get_favicon"}],["$","$L5","4",{}]]}]}]}],["$","meta",null,{"name":"next-size-adjust","content":""}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"5rDiFx0t_mOGYmV_8kSkw"}
|
||||
5:I[27201,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"IconMark"]
|
||||
0:{"rsc":["$","$1","h",{"children":[null,["$","$L2",null,{"children":[["$","meta","0",{"charSet":"utf-8"}],["$","meta","1",{"name":"viewport","content":"width=device-width, initial-scale=1"}]]}],["$","div",null,{"hidden":true,"children":["$","$L3",null,{"children":["$","$4",null,{"name":"Next.Metadata","children":[["$","title","0",{"children":"LiteLLM Dashboard"}],["$","meta","1",{"name":"description","content":"LiteLLM Proxy Admin UI"}],["$","link","2",{"rel":"icon","href":"/favicon.ico?favicon.0~dgapwhi~75y.ico","sizes":"48x48","type":"image/x-icon"}],["$","link","3",{"rel":"icon","href":"/get_favicon"}],["$","$L5","4",{}]]}]}]}],["$","meta",null,{"name":"next-size-adjust","content":""}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"KYqiq5stbD-H4YcZ-6OuP"}
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
1:"$Sreact.fragment"
|
||||
2:I[867271,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
3:I[71195,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
4:I[557951,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"AuthProvider"]
|
||||
5:I[339756,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
6:I[837457,["/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
2:I[867271,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
3:I[71195,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
4:I[557951,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"AuthProvider"]
|
||||
5:I[339756,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
6:I[837457,["/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js"],"default"]
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/05qmwjqau64bz.css","style"]
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/0i77.0u.82o9u.css","style"]
|
||||
0:{"rsc":["$","$1","c",{"children":[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/05qmwjqau64bz.css","precedence":"next"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/0i77.0u.82o9u.css","precedence":"next"}],["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/0n.a~e5dwfnkn.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/0.4.bbjx7y007.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js","async":true}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"inter_5972bc34-module__OU16Qa__className","children":["$","$L2",null,{"children":["$","$L3",null,{"children":["$","$L4",null,{"children":["$","$L5",null,{"parallelRouterKey":"children","template":["$","$L6",null,{}],"notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]]}]}]}]}]}]}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"5rDiFx0t_mOGYmV_8kSkw"}
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/075sund.-mh4~.css","style"]
|
||||
0:{"rsc":["$","$1","c",{"children":[[["$","link","0",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/05qmwjqau64bz.css","precedence":"next"}],["$","link","1",{"rel":"stylesheet","href":"/litellm-asset-prefix/_next/static/chunks/075sund.-mh4~.css","precedence":"next"}],["$","script","script-0",{"src":"/litellm-asset-prefix/_next/static/chunks/08yy42xvwaak6.js","async":true}],["$","script","script-1",{"src":"/litellm-asset-prefix/_next/static/chunks/0e9hs7onyj28m.js","async":true}],["$","script","script-2",{"src":"/litellm-asset-prefix/_next/static/chunks/0pidya1qvuvx8.js","async":true}]],["$","html",null,{"lang":"en","children":["$","body",null,{"className":"inter_5972bc34-module__OU16Qa__className","children":["$","$L2",null,{"children":["$","$L3",null,{"children":["$","$L4",null,{"children":["$","$L5",null,{"parallelRouterKey":"children","template":["$","$L6",null,{}],"notFound":[[["$","title",null,{"children":"404: This page could not be found."}],["$","div",null,{"style":{"fontFamily":"system-ui,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif,\"Apple Color Emoji\",\"Segoe UI Emoji\"","height":"100vh","textAlign":"center","display":"flex","flexDirection":"column","alignItems":"center","justifyContent":"center"},"children":["$","div",null,{"children":[["$","style",null,{"dangerouslySetInnerHTML":{"__html":"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}"}}],["$","h1",null,{"className":"next-error-h1","style":{"display":"inline-block","margin":"0 20px 0 0","padding":"0 23px 0 0","fontSize":24,"fontWeight":500,"verticalAlign":"top","lineHeight":"49px"},"children":404}],["$","div",null,{"style":{"display":"inline-block"},"children":["$","h2",null,{"style":{"fontSize":14,"fontWeight":400,"lineHeight":"49px","margin":0},"children":"This page could not be found."}]}]]}]}]],[]]}]}]}]}]}]}]]}],"isPartial":false,"staleTime":300,"varyParams":null,"buildId":"KYqiq5stbD-H4YcZ-6OuP"}
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
:HL["/litellm-asset-prefix/_next/static/chunks/05qmwjqau64bz.css","style"]
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/0i77.0u.82o9u.css","style"]
|
||||
:HL["/litellm-asset-prefix/_next/static/chunks/075sund.-mh4~.css","style"]
|
||||
:HL["/litellm-asset-prefix/_next/static/media/83afe278b6a6bb3c-s.p.0q-301v4kxxnr.woff2","font",{"crossOrigin":"","type":"font/woff2"}]
|
||||
0:{"tree":{"name":"","param":null,"prefetchHints":16,"slots":{"children":{"name":"(dashboard)","param":null,"prefetchHints":0,"slots":{"children":{"name":"__PAGE__","param":null,"prefetchHints":0,"slots":null}}}}},"staleTime":300,"buildId":"5rDiFx0t_mOGYmV_8kSkw"}
|
||||
0:{"tree":{"name":"","param":null,"prefetchHints":16,"slots":{"children":{"name":"(dashboard)","param":null,"prefetchHints":0,"slots":{"children":{"name":"__PAGE__","param":null,"prefetchHints":0,"slots":null}}}}},"staleTime":300,"buildId":"KYqiq5stbD-H4YcZ-6OuP"}
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -1 +0,0 @@
|
|||
(globalThis.TURBOPACK||(globalThis.TURBOPACK=[])).push(["object"==typeof document?document.currentScript:void 0,526612,e=>{"use strict";var t=e.i(843476),u=e.i(846835),s=e.i(135214);e.s(["default",0,function(){let{accessToken:e,userRole:i,premiumUser:o}=(0,s.default)();return(0,t.jsx)(u.default,{userRole:i??"",accessToken:e,premiumUser:o??!1})}])}]);
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue